{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Installs","metadata":{}},{"cell_type":"code","source":"!pip install -qq duckdb rdkit scikit-fingerprints","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:25:10.219485Z","iopub.execute_input":"2024-06-26T10:25:10.219913Z","iopub.status.idle":"2024-06-26T10:25:29.252615Z","shell.execute_reply.started":"2024-06-26T10:25:10.219883Z","shell.execute_reply":"2024-06-26T10:25:29.251004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import duckdb\nimport pandas as pd\nfrom rdkit import Chem\nfrom rdkit.Chem import AllChem\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder\nfrom skfp.fingerprints import ECFPFingerprint\nfrom tqdm import tqdm\nimport numpy as np\nimport joblib\nimport os\nimport gc","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:25:29.255371Z","iopub.execute_input":"2024-06-26T10:25:29.255781Z","iopub.status.idle":"2024-06-26T10:25:45.869015Z","shell.execute_reply.started":"2024-06-26T10:25:29.255731Z","shell.execute_reply":"2024-06-26T10:25:45.867827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to free memory once a variable is no longer required\ndef free_memory(*vars):\n    for var in vars:\n        del var\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:25:45.870508Z","iopub.execute_input":"2024-06-26T10:25:45.871310Z","iopub.status.idle":"2024-06-26T10:25:45.876735Z","shell.execute_reply.started":"2024-06-26T10:25:45.871278Z","shell.execute_reply":"2024-06-26T10:25:45.875653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# paths\ntrain_path = '/kaggle/input/leash-predict-chemical-bindings/train.parquet'\ntest_path = '/kaggle/input/leash-predict-chemical-bindings/test.parquet'","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:25:45.879787Z","iopub.execute_input":"2024-06-26T10:25:45.880226Z","iopub.status.idle":"2024-06-26T10:25:45.897185Z","shell.execute_reply.started":"2024-06-26T10:25:45.880191Z","shell.execute_reply":"2024-06-26T10:25:45.895954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"con = duckdb.connect()\n\n# Query all rows where binds = 1\nbinds_1_df = con.query(f\"\"\"\n                        SELECT molecule_smiles, protein_name, binds\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                       \"\"\").df()\n\n# Count the number of rows where binds = 1\nnum_binds_1 = len(binds_1_df)\n\n# Query the same number of rows where binds = 0\nbinds_0_df = con.query(f\"\"\"\n                        SELECT molecule_smiles, protein_name, binds\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT {num_binds_1}\n                       \"\"\").df()\n\n# Combine the two DataFrames\ndf = pd.concat([binds_1_df, binds_0_df])\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:25:45.898932Z","iopub.execute_input":"2024-06-26T10:25:45.899736Z","iopub.status.idle":"2024-06-26T10:27:06.950986Z","shell.execute_reply.started":"2024-06-26T10:25:45.899697Z","shell.execute_reply":"2024-06-26T10:27:06.949255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['protein_name'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:29:08.283713Z","iopub.execute_input":"2024-06-26T10:29:08.284142Z","iopub.status.idle":"2024-06-26T10:29:08.619421Z","shell.execute_reply.started":"2024-06-26T10:29:08.284110Z","shell.execute_reply":"2024-06-26T10:29:08.618230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the mapping from protein names to integers\nprotein_mapping = {\n    'HSA': 0,\n    'sEH': 1,\n    'BRD4': 2\n}","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:43:26.258993Z","iopub.execute_input":"2024-06-26T10:43:26.259468Z","iopub.status.idle":"2024-06-26T10:43:26.265530Z","shell.execute_reply.started":"2024-06-26T10:43:26.259434Z","shell.execute_reply":"2024-06-26T10:43:26.264247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['protein_name'] = df['protein_name'].map(protein_mapping)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:43:27.883685Z","iopub.execute_input":"2024-06-26T10:43:27.884083Z","iopub.status.idle":"2024-06-26T10:43:28.202466Z","shell.execute_reply.started":"2024-06-26T10:43:27.884048Z","shell.execute_reply":"2024-06-26T10:43:28.201148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:43:35.230500Z","iopub.execute_input":"2024-06-26T10:43:35.230928Z","iopub.status.idle":"2024-06-26T10:43:35.242617Z","shell.execute_reply.started":"2024-06-26T10:43:35.230900Z","shell.execute_reply":"2024-06-26T10:43:35.241375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to efficiently calculate ecfps using scikit-fingerprint\nELEMENTS_PER_WORKER = 100_000\n\ndef calculate_fp(x):\n#     if fp_name == 'ecfp':\n    fp_transformer =  ECFPFingerprint(fp_size=1024, radius=3, n_jobs=-1)\n#     elif fp_name == 'ap':\n#         fp_transformer = fp_ap_transformer\n#     elif fp_name == 'maccs':\n#         fp_transformer = fp_maccs_transformer\n#     else:\n#         raise Exception('Wrong fingerprint name!')\n    \n    middle_parts = []\n    k_splits = x.shape[0] // ELEMENTS_PER_WORKER\n        \n    for i in tqdm(range(k_splits)):\n        middle_parts.append(fp_transformer.transform(x[i * ELEMENTS_PER_WORKER: (i + 1) * ELEMENTS_PER_WORKER]))\n        \n    if x.shape[0] % ELEMENTS_PER_WORKER > 0:   \n        middle_parts.append(fp_transformer.transform(x[k_splits * ELEMENTS_PER_WORKER:]))\n    \n    return np.concatenate(middle_parts)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:43:47.437257Z","iopub.execute_input":"2024-06-26T10:43:47.437681Z","iopub.status.idle":"2024-06-26T10:43:47.445322Z","shell.execute_reply.started":"2024-06-26T10:43:47.437649Z","shell.execute_reply":"2024-06-26T10:43:47.444087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\necfp_train = calculate_fp(df['molecule_smiles'])\n","metadata":{"execution":{"iopub.status.busy":"2024-06-26T10:43:51.141251Z","iopub.execute_input":"2024-06-26T10:43:51.141668Z","iopub.status.idle":"2024-06-26T10:59:00.565352Z","shell.execute_reply.started":"2024-06-26T10:43:51.141635Z","shell.execute_reply":"2024-06-26T10:59:00.563713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the target variable\ny = df['binds'].values.astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:00:37.528365Z","iopub.execute_input":"2024-06-26T11:00:37.528898Z","iopub.status.idle":"2024-06-26T11:00:37.543285Z","shell.execute_reply.started":"2024-06-26T11:00:37.528858Z","shell.execute_reply":"2024-06-26T11:00:37.542078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_onehot = df['protein_name'].values","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:10:00.440017Z","iopub.execute_input":"2024-06-26T11:10:00.440458Z","iopub.status.idle":"2024-06-26T11:10:00.446610Z","shell.execute_reply.started":"2024-06-26T11:10:00.440427Z","shell.execute_reply":"2024-06-26T11:10:00.445072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # One-hot encode the protein_name\n# onehot_encoder = OneHotEncoder(sparse_output=False)\n# protein_onehot = onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2024-06-21T10:37:45.476121Z","iopub.execute_input":"2024-06-21T10:37:45.476868Z","iopub.status.idle":"2024-06-21T10:37:46.658360Z","shell.execute_reply.started":"2024-06-21T10:37:45.476830Z","shell.execute_reply":"2024-06-21T10:37:46.657073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:12:55.497797Z","iopub.execute_input":"2024-06-26T11:12:55.498172Z","iopub.status.idle":"2024-06-26T11:12:55.505663Z","shell.execute_reply.started":"2024-06-26T11:12:55.498146Z","shell.execute_reply":"2024-06-26T11:12:55.504504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"free_memory(df)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:10:09.706062Z","iopub.execute_input":"2024-06-26T11:10:09.706471Z","iopub.status.idle":"2024-06-26T11:10:10.086969Z","shell.execute_reply.started":"2024-06-26T11:10:09.706440Z","shell.execute_reply":"2024-06-26T11:10:10.085350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert protein_onehot to np.uint8\n# protein_onehot = protein_onehot.astype(np.uint8)\necfp_train = ecfp_train.astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:10:18.307740Z","iopub.execute_input":"2024-06-26T11:10:18.308232Z","iopub.status.idle":"2024-06-26T11:10:21.719613Z","shell.execute_reply.started":"2024-06-26T11:10:18.308198Z","shell.execute_reply":"2024-06-26T11:10:21.718265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def generate_combined_array(mode, ecfp_array, protein_onehot_array, chunk_size):\n#     # Determine the shapes of the arrays\n#     num_samples = ecfp_array.shape[0]\n#     ecfp_dim = ecfp_array.shape[1]\n#     protein_onehot_dim = protein_onehot_array.shape[1]\n#     if mode == 'train':\n#         combined_shape = (num_samples, ecfp_dim + protein_onehot_dim + 1)  # +1 for the target variable\n#     elif mode == 'test':\n#         combined_shape = (num_samples, ecfp_dim + protein_onehot_dim)\n#     combined_filename = f'{mode}.dat'\n#     combined_array = np.memmap(combined_filename, dtype=np.uint8, mode='w+', shape=combined_shape)\n#     return combined_array, num_samples","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:10:26.132147Z","iopub.execute_input":"2024-06-26T11:10:26.132554Z","iopub.status.idle":"2024-06-26T11:10:26.140774Z","shell.execute_reply.started":"2024-06-26T11:10:26.132523Z","shell.execute_reply":"2024-06-26T11:10:26.139388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_combined_array(mode, ecfp_array,chunk_size):\n    # Determine the shapes of the arrays\n    num_samples = ecfp_array.shape[0]\n    ecfp_dim = ecfp_array.shape[1]\n    if mode == 'train':\n        combined_shape = (num_samples, ecfp_dim + 1 + 1)  # +1 for the target variable\n    elif mode == 'test':\n        combined_shape = (num_samples, ecfp_dim + 1)\n    combined_filename = f'{mode}.dat'\n    combined_array = np.memmap(combined_filename, dtype=np.uint8, mode='w+', shape=combined_shape)\n    return combined_array, num_samples","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:17:35.233383Z","iopub.execute_input":"2024-06-26T11:17:35.234685Z","iopub.status.idle":"2024-06-26T11:17:35.242467Z","shell.execute_reply.started":"2024-06-26T11:17:35.234641Z","shell.execute_reply":"2024-06-26T11:17:35.241077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nchunk_size = 100_000\n\ncombined_array, num_samples = generate_combined_array('train', ecfp_train, chunk_size)\n# Process and combine arrays in chunks\nfor start in range(0, num_samples, chunk_size):\n    end = min(start + chunk_size, num_samples)\n    combined_chunk = np.hstack((ecfp_train[start:end], protein_onehot[start:end].reshape(-1, 1), y[start:end].reshape(-1, 1)))\n    combined_array[start:end] = combined_chunk\n\n# Flush changes to disk\ncombined_array.flush()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:17:51.787738Z","iopub.execute_input":"2024-06-26T11:17:51.788184Z","iopub.status.idle":"2024-06-26T11:18:19.837514Z","shell.execute_reply.started":"2024-06-26T11:17:51.788152Z","shell.execute_reply":"2024-06-26T11:18:19.836229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"free_memory(ecfp_train, y, combined_array)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:18:37.230339Z","iopub.execute_input":"2024-06-26T11:18:37.230822Z","iopub.status.idle":"2024-06-26T11:18:37.410180Z","shell.execute_reply.started":"2024-06-26T11:18:37.230792Z","shell.execute_reply":"2024-06-26T11:18:37.408665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined_shape = (3179812, 1026) \ncombined_array = np.memmap('/kaggle/working/train.dat', dtype=np.uint8, mode='r', shape=combined_shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:18:54.425214Z","iopub.execute_input":"2024-06-26T11:18:54.425669Z","iopub.status.idle":"2024-06-26T11:18:54.532175Z","shell.execute_reply.started":"2024-06-26T11:18:54.425636Z","shell.execute_reply":"2024-06-26T11:18:54.530945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Determine the number of feature columns (excluding the target column)\nnum_feature_columns = combined_shape[1] - 1","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:18:57.345767Z","iopub.execute_input":"2024-06-26T11:18:57.346889Z","iopub.status.idle":"2024-06-26T11:18:57.351929Z","shell.execute_reply.started":"2024-06-26T11:18:57.346846Z","shell.execute_reply":"2024-06-26T11:18:57.350619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate X and y\nX = combined_array[:, :num_feature_columns]  # All rows, all columns except the last\ny = combined_array[:, num_feature_columns]   # All rows, only the last column","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:19:07.918995Z","iopub.execute_input":"2024-06-26T11:19:07.919997Z","iopub.status.idle":"2024-06-26T11:19:07.925336Z","shell.execute_reply.started":"2024-06-26T11:19:07.919955Z","shell.execute_reply":"2024-06-26T11:19:07.924111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Define the Random Forest model with controlled memory usage\nrf_model = RandomForestClassifier(n_estimators=10, max_depth=10, n_jobs=-1, random_state=42)\n\n# Split the data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nrf_model.fit(X_train, y_train)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:19:27.200710Z","iopub.execute_input":"2024-06-26T11:19:27.202043Z","iopub.status.idle":"2024-06-26T11:22:28.384472Z","shell.execute_reply.started":"2024-06-26T11:19:27.201984Z","shell.execute_reply":"2024-06-26T11:22:28.383142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"free_memory(combined_array, X, y)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:24:02.337086Z","iopub.execute_input":"2024-06-26T11:24:02.337535Z","iopub.status.idle":"2024-06-26T11:24:02.530652Z","shell.execute_reply.started":"2024-06-26T11:24:02.337505Z","shell.execute_reply":"2024-06-26T11:24:02.529156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Make predictions on the test set\ny_pred_proba = rf_model.predict_proba(X_test)[:, 1]  # Probability of the positive class\n\n# Calculate the mean average precision\nmap_score = average_precision_score(y_test, y_pred_proba)\nprint(f\"Mean Average Precision (mAP): {map_score:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:24:03.366773Z","iopub.execute_input":"2024-06-26T11:24:03.367191Z","iopub.status.idle":"2024-06-26T11:24:06.776996Z","shell.execute_reply.started":"2024-06-26T11:24:03.367159Z","shell.execute_reply":"2024-06-26T11:24:06.773389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_filename = 'random_forest_model.joblib'\nprint(f\"Saving the model to {model_filename}...\")\njoblib.dump(rf_model, model_filename)\nprint(\"Model saved successfully.\")","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:24:16.919141Z","iopub.execute_input":"2024-06-26T11:24:16.919612Z","iopub.status.idle":"2024-06-26T11:24:16.938969Z","shell.execute_reply.started":"2024-06-26T11:24:16.919567Z","shell.execute_reply":"2024-06-26T11:24:16.937672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"free_memory(X_train, X_test, y_train, y_test, y_pred_proba)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:24:18.235173Z","iopub.execute_input":"2024-06-26T11:24:18.235562Z","iopub.status.idle":"2024-06-26T11:24:18.413618Z","shell.execute_reply.started":"2024-06-26T11:24:18.235534Z","shell.execute_reply":"2024-06-26T11:24:18.412216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"con = duckdb.connect()\n\n# Query all rows where binds = 1\ntest_df = con.query(f\"\"\"\n                        SELECT id, molecule_smiles,\tprotein_name,\n                        FROM parquet_scan('{test_path}')\n                       \"\"\").df()\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:24:21.967726Z","iopub.execute_input":"2024-06-26T11:24:21.968987Z","iopub.status.idle":"2024-06-26T11:24:23.471191Z","shell.execute_reply.started":"2024-06-26T11:24:21.968938Z","shell.execute_reply":"2024-06-26T11:24:23.470071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:24:26.153746Z","iopub.execute_input":"2024-06-26T11:24:26.154223Z","iopub.status.idle":"2024-06-26T11:24:26.167416Z","shell.execute_reply.started":"2024-06-26T11:24:26.154189Z","shell.execute_reply":"2024-06-26T11:24:26.166106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\necfp_test = calculate_fp(test_df['molecule_smiles'])","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:25:45.158367Z","iopub.execute_input":"2024-06-26T11:25:45.159552Z","iopub.status.idle":"2024-06-26T11:34:10.746888Z","shell.execute_reply.started":"2024-06-26T11:25:45.159509Z","shell.execute_reply":"2024-06-26T11:34:10.745337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the mapping from protein names to integers\nprotein_mapping = {\n    'HSA': 0,\n    'sEH': 1,\n    'BRD4': 2\n}\ndf['protein_name'] = df['protein_name'].map(protein_mapping)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:34:10.750213Z","iopub.execute_input":"2024-06-26T11:34:10.750615Z","iopub.status.idle":"2024-06-26T11:34:11.119450Z","shell.execute_reply.started":"2024-06-26T11:34:10.750563Z","shell.execute_reply":"2024-06-26T11:34:11.118036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ecfp_test = ecfp_test.astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:34:11.121252Z","iopub.execute_input":"2024-06-26T11:34:11.121636Z","iopub.status.idle":"2024-06-26T11:34:11.945339Z","shell.execute_reply.started":"2024-06-26T11:34:11.121605Z","shell.execute_reply":"2024-06-26T11:34:11.943975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nchunk_size = 100_000\ncombined_array, num_samples = generate_combined_array(mode='test', ecfp_array=ecfp_test, chunk_size=100_000)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:36:33.955190Z","iopub.execute_input":"2024-06-26T11:36:33.956547Z","iopub.status.idle":"2024-06-26T11:36:33.965209Z","shell.execute_reply.started":"2024-06-26T11:36:33.956507Z","shell.execute_reply":"2024-06-26T11:36:33.963803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Process and combine arrays in chunks\nfor start in range(0, num_samples, chunk_size):\n    end = min(start + chunk_size, num_samples)\n    combined_chunk = np.hstack((ecfp_test[start:end], protein_onehot[start:end].reshape(-1, 1)))\n    combined_array[start:end] = combined_chunk\n\n# Flush changes to disk\ncombined_array.flush()","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:36:41.744336Z","iopub.execute_input":"2024-06-26T11:36:41.744870Z","iopub.status.idle":"2024-06-26T11:36:58.992107Z","shell.execute_reply.started":"2024-06-26T11:36:41.744832Z","shell.execute_reply":"2024-06-26T11:36:58.990514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined_array.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:38:09.117639Z","iopub.execute_input":"2024-06-26T11:38:09.118148Z","iopub.status.idle":"2024-06-26T11:38:09.126340Z","shell.execute_reply.started":"2024-06-26T11:38:09.118116Z","shell.execute_reply":"2024-06-26T11:38:09.124974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Predict the probabilities\nprobabilities = rf_model.predict_proba(combined_array)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:38:13.162563Z","iopub.execute_input":"2024-06-26T11:38:13.163874Z","iopub.status.idle":"2024-06-26T11:38:21.189265Z","shell.execute_reply.started":"2024-06-26T11:38:13.163824Z","shell.execute_reply":"2024-06-26T11:38:21.187645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\noutput_file = 'submission.csv'\n# Create a DataFrame with 'id' and 'probability' columns\noutput_df = pd.DataFrame({'id': test_df['id'], 'binds': probabilities})\n\n# Save the output DataFrame to a CSV file\noutput_df.to_csv(output_file, index=False, mode='a', header=not os.path.exists(output_file))\n","metadata":{"execution":{"iopub.status.busy":"2024-06-26T11:38:30.509955Z","iopub.execute_input":"2024-06-26T11:38:30.510424Z","iopub.status.idle":"2024-06-26T11:38:34.430212Z","shell.execute_reply.started":"2024-06-26T11:38:30.510390Z","shell.execute_reply":"2024-06-26T11:38:34.428746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}