{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30618,"isInternetEnabled":true,"language":"r","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Some good solutions are already proposed to convert source train data into something more feasible to deal with.\n\nLet me share possible the simplest approach for direct saving all train data in wide format without any OOM issues using `duckdb` package in R (Python package should work the same way). Resulting file can be queried for smaller subsets or handle in RAM on Kaggle.","metadata":{}},{"cell_type":"code","source":"install.packages(\"arrow\")\n# install.packages(\"duckdb\")\n\nlibrary(arrow)\nlibrary(duckdb)\nlibrary(dplyr)\n\ncon <- dbConnect(duckdb::duckdb())\ntrain <- open_dataset(\"/kaggle/input/leash-BELKA/train.parquet\") %>%\n  select(molecule_smiles, protein_name, binds) %>%\n  to_duckdb(table_name = \"train\", con = con)\n\ndbSendQuery(\n  con, \n  \"COPY(\n  SELECT\n    molecule_smiles , \n    max(case when protein_name = 'BRD4' then binds else null end) as BRD4, \n    max(case when protein_name = 'HSA' then binds else null end) as HSA, \n    max(case when protein_name = 'sEH' then binds else null end) as sEH\n  FROM train\n  GROUP BY molecule_smiles) \n  TO 'train_wide.parquet' \n  (field_ids 'auto')\"\n)\ngc()","metadata":{"execution":{"iopub.status.busy":"2024-04-30T08:31:22.823050Z","iopub.execute_input":"2024-04-30T08:31:22.825328Z","iopub.status.idle":"2024-04-30T08:39:57.903968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt <- open_dataset(\"train_wide.parquet\")\ncollect(head(dt)) # start fresh session to load the entire dataset","metadata":{"execution":{"iopub.status.busy":"2024-04-30T08:42:00.168761Z","iopub.execute_input":"2024-04-30T08:42:00.170424Z"},"trusted":true},"execution_count":null,"outputs":[]}]}