{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%pip install git+https://github.com/dls5-omics/multimolecule@develop --quiet","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-06T21:09:08.019642Z","iopub.execute_input":"2025-03-06T21:09:08.020362Z","iopub.status.idle":"2025-03-06T21:09:27.532884Z","shell.execute_reply.started":"2025-03-06T21:09:08.02027Z","shell.execute_reply":"2025-03-06T21:09:27.531067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import danling as dl\nimport multimolecule as mm\nimport pandas as pd\nfrom chanfig import Config\nfrom multimolecule import MultiMoleculeConfig, MultiMoleculeRunner","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T21:09:27.535807Z","iopub.execute_input":"2025-03-06T21:09:27.536259Z","iopub.status.idle":"2025-03-06T21:09:38.383241Z","shell.execute_reply.started":"2025-03-06T21:09:27.536219Z","shell.execute_reply":"2025-03-06T21:09:38.381968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_data(df):\n    df['ID'] = df['ID'].str.split('_').str[0] + '_' + df['ID'].str.split('_').str[1]\n    grouped = df.groupby('ID').agg({\n        'resname': lambda x: ''.join(x),\n        'x_1': lambda x: list(x),\n        'y_1': lambda x: list(x),\n        'z_1': lambda x: list(x),\n    }).reset_index()\n    grouped.set_index(\"ID\")\n    return grouped.rename(columns={'resname': 'sequence'})\n\nprocess_data(dl.load(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")).to_json(\"train.json\", index=False)\nprocess_data(dl.load(\"/kaggle/input/stanford-rna-3d-folding/validation_labels.csv\")).to_json(\"validation.json\", index=False)\n# process_data(dl.load(\"/kaggle/input/stanford-rna-3d-folding/test_labels.csv\")).to_json(\"test.json\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T21:09:38.385349Z","iopub.execute_input":"2025-03-06T21:09:38.386578Z","iopub.status.idle":"2025-03-06T21:09:40.66613Z","shell.execute_reply.started":"2025-03-06T21:09:38.38651Z","shell.execute_reply":"2025-03-06T21:09:40.664737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = Config(train=\"train.json\", validation=\"validation.json\")\nconfig = MultiMoleculeConfig(data=data, pretrained=\"multimolecule/ernierna\")\nconfig.parse()  # Parse must be called because it also handles post-processing\nrunner = MultiMoleculeRunner(config)\nrunner.train()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T21:09:40.667433Z","iopub.execute_input":"2025-03-06T21:09:40.667778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}