{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":67356,"databundleVersionId":8006601,"isSourceIdPinned":false}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore', category=DeprecationWarning)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T10:00:48.099155Z","iopub.execute_input":"2026-03-15T10:00:48.099807Z","iopub.status.idle":"2026-03-15T10:00:48.104607Z","shell.execute_reply.started":"2026-03-15T10:00:48.099777Z","shell.execute_reply":"2026-03-15T10:00:48.103540Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2026-03-15T10:00:48.106439Z","iopub.execute_input":"2026-03-15T10:00:48.106768Z","iopub.status.idle":"2026-03-15T10:00:52.332198Z","shell.execute_reply.started":"2026-03-15T10:00:48.106743Z","shell.execute_reply":"2026-03-15T10:00:52.330998Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install duckdb","metadata":{"execution":{"iopub.status.busy":"2026-03-15T10:00:52.334028Z","iopub.execute_input":"2026-03-15T10:00:52.335338Z","iopub.status.idle":"2026-03-15T10:00:56.453271Z","shell.execute_reply.started":"2026-03-15T10:00:52.335288Z","shell.execute_reply":"2026-03-15T10:00:56.452152Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import duckdb\nimport pandas as pd\n\ntrain_path = '/kaggle/input/competitions/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/competitions/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\ndf = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 500)\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 500)\"\"\").df()\n\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2026-03-15T10:00:56.456266Z","iopub.execute_input":"2026-03-15T10:00:56.456677Z","iopub.status.idle":"2026-03-15T10:01:22.720938Z","shell.execute_reply.started":"2026-03-15T10:00:56.456642Z","shell.execute_reply":"2026-03-15T10:01:22.719939Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-15T10:01:22.722345Z","iopub.execute_input":"2026-03-15T10:01:22.722736Z","iopub.status.idle":"2026-03-15T10:01:22.736292Z","shell.execute_reply.started":"2026-03-15T10:01:22.722684Z","shell.execute_reply":"2026-03-15T10:01:22.735265Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from rdkit import Chem\n# FIX: Use rdFingerprintGenerator instead of deprecated AllChem.GetMorganFingerprintAsBitVect\nfrom rdkit.Chem import rdFingerprintGenerator\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder\nimport numpy as np\n\n# Convert SMILES to RDKit molecules\ndf['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)\n\n# FIX: Use rdFingerprintGenerator (replaces deprecated GetMorganFingerprintAsBitVect)\n# Returns a numpy array directly, which works correctly with concatenation later\nmfpgen = rdFingerprintGenerator.GetMorganGenerator(radius=2, fpSize=1024)\n\ndef generate_ecfp(molecule):\n    if molecule is None:\n        return None\n    return mfpgen.GetFingerprintAsNumPy(molecule)\n\ndf['ecfp'] = df['molecule'].apply(generate_ecfp)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T10:01:22.737630Z","iopub.execute_input":"2026-03-15T10:01:22.737937Z","iopub.status.idle":"2026-03-15T10:01:23.224734Z","shell.execute_reply.started":"2026-03-15T10:01:22.737911Z","shell.execute_reply":"2026-03-15T10:01:23.223391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder\n\n# One-hot encode the protein_name\nonehot_encoder = OneHotEncoder(sparse_output=False)\nprotein_onehot = onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1, 1))\n\n# Combine ECFPs and one-hot encoded protein_name\nX = np.array([\n    np.concatenate([ecfp, protein])\n    for ecfp, protein in zip(df['ecfp'].tolist(), protein_onehot.tolist())\n])\ny = df['binds'].tolist()\n\n# Split the data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Create and train the random forest model\nrf_model = RandomForestClassifier(n_estimators=100, random_state=42)\nrf_model.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred_proba = rf_model.predict_proba(X_test)[:, 1]\n\n# Calculate the mean average precision\nmap_score = average_precision_score(y_test, y_pred_proba)\nprint(f\"Mean Average Precision (mAP): {map_score:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2026-03-15T10:01:23.226065Z","iopub.execute_input":"2026-03-15T10:01:23.226439Z","iopub.status.idle":"2026-03-15T10:01:23.731659Z","shell.execute_reply.started":"2026-03-15T10:01:23.226402Z","shell.execute_reply":"2026-03-15T10:01:23.730536Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\n\ntest_file = '/kaggle/input/competitions/leash-BELKA/test.csv'\noutput_file = 'submission.csv'\nchunksize = 100000\n\ntotal_rows = sum(1 for _ in open(test_file)) - 1\n\nwith tqdm(total=total_rows, desc=\"Processing\", unit=\"rows\") as pbar:\n    for df_test in pd.read_csv(test_file, chunksize=chunksize):\n\n        df_test['molecule'] = df_test['molecule_smiles'].apply(Chem.MolFromSmiles)\n        df_test['ecfp'] = df_test['molecule'].apply(generate_ecfp)\n\n        protein_onehot = onehot_encoder.transform(df_test['protein_name'].values.reshape(-1, 1))\n\n        X_test_chunk = np.array([\n            np.concatenate([ecfp, protein])\n            for ecfp, protein in zip(df_test['ecfp'].tolist(), protein_onehot.tolist())\n        ])\n\n        probabilities = rf_model.predict_proba(X_test_chunk)[:, 1]\n\n        output_df = pd.DataFrame({'id': df_test['id'], 'binds': probabilities})\n        output_df.to_csv(output_file, index=False, mode='a', header=not os.path.exists(output_file))\n\n        pbar.update(len(df_test))\n\nprint(f\"✅ Done! Submission saved to {output_file}\")","metadata":{"execution":{"iopub.status.busy":"2026-03-15T10:01:23.732947Z","iopub.execute_input":"2026-03-15T10:01:23.733676Z","iopub.status.idle":"2026-03-15T10:15:05.998029Z","shell.execute_reply.started":"2026-03-15T10:01:23.733644Z","shell.execute_reply":"2026-03-15T10:15:05.996914Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}