{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Overview\n\nSo, our aims to predict the three-dimensional (3D) structure of RNA molecules from their nucleotide sequences.\n\n[Visualisation](https://www.kaggle.com/code/asarvazyan/interactive-3d-sequence-visualization) \n\n[Features ](https://www.kaggle.com/code/dantheshark/rna-3d-folding-understand-data)","metadata":{}},{"cell_type":"code","source":"# packages\n\n# standard\nimport numpy as np\nimport pandas as pd\nimport time\n\nimport nltk\nfrom sklearn.feature_extraction.text import CountVectorizer,TfidfVectorizer\nfrom sklearn.decomposition import TruncatedSVD\n\n# plots\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport seaborn as sns\n\nimport plotly.graph_objects as go\n\n# warning handling\nimport warnings\nwarnings.filterwarnings('ignore', category=FutureWarning)  \nwarnings.filterwarnings('ignore', category=RuntimeWarning)\n\n# configs\npd.set_option('display.max_columns', 300)\npd.set_option('display.max_rows', 150)\n\ndefault_color_1 = 'darkblue'\nset_plot = False\n\n\n\nfrom colorama import Style, Fore\nblk = Style.BRIGHT + Fore.BLACK\nred = Style.BRIGHT + Fore.RED\nblu = Style.BRIGHT + Fore.BLUE\nclr = Style.RESET_ALL\n\n# load data\ndf_train = pd.read_csv('../input/stanford-rna-3d-folding/train_labels.csv')\ndf_valid = pd.read_csv('../input/stanford-rna-3d-folding/validation_labels.csv')\ndf_train_seq = pd.read_csv('../input/stanford-rna-3d-folding/train_sequences.csv')\ndf_valid_seq = pd.read_csv('../input/stanford-rna-3d-folding/validation_sequences.csv')\ndf_test = pd.read_csv('../input/stanford-rna-3d-folding/test_sequences.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:22:46.304670Z","iopub.execute_input":"2025-12-01T10:22:46.304951Z","iopub.status.idle":"2025-12-01T10:22:49.215303Z","shell.execute_reply.started":"2025-12-01T10:22:46.304930Z","shell.execute_reply":"2025-12-01T10:22:49.214464Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data preprocessing","metadata":{}},{"cell_type":"code","source":"# ------------------\n# Train data\n# ------------------\n\n#df_train_seq[\"all_sequences_str\"] = df_train_seq[\"all_sequences\"].astype(str)\ndf_train_seq['sequence_id'] = df_train_seq[\"target_id\"]\ndf_train_seq[\"sequence_num\"] = df_train_seq[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"|\")[0:1])[1:])\ndf_train_seq[\"sequence_group\"] = df_train_seq[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"_\")[0:1])[1:])\ndf_train_seq[\"sequence_group_num\"] = df_train_seq[\"sequence_num\"].apply(lambda x: \"_\".join(x.split(\"_\")[1:2]))\n\n# for nan in all_sequences handle  sequence_group    df_train_seq[df_train_seq[\"all_sequences\"].isna()]\n\ndf_train_seq.loc[df_train_seq.target_id == '2ZJQ_Y', 'sequence_num'] = '2ZJQ_1'\ndf_train_seq.loc[df_train_seq.target_id == '2ZJQ_X', 'sequence_num'] = '2ZJQ_1'\ndf_train_seq.loc[df_train_seq.target_id == '4V65_A1', 'sequence_num'] = '4V65_1'\ndf_train_seq.loc[df_train_seq.target_id == '4V65_BB', 'sequence_num'] = '4V65_1'\ndf_train_seq.loc[df_train_seq.target_id == '4V5F_CA', 'sequence_num'] = '4V5F_1'\n\ndf_train_seq.loc[df_train_seq.target_id == '2ZJQ_Y', 'sequence_group'] = '2ZJQ_Y'\ndf_train_seq.loc[df_train_seq.target_id == '2ZJQ_X', 'sequence_group'] = '2ZJQ_X'\ndf_train_seq.loc[df_train_seq.target_id == '4V65_A1', 'sequence_group'] = '4V65_A1'\ndf_train_seq.loc[df_train_seq.target_id == '4V65_BB', 'sequence_group'] = '4V65_BB'\ndf_train_seq.loc[df_train_seq.target_id == '4V5F_CA', 'sequence_group'] = '4V5F_CA'\n\ndf_train_seq.loc[df_train_seq.target_id == '2ZJQ_Y', 'sequence_group_num'] = 1\ndf_train_seq.loc[df_train_seq.target_id == '2ZJQ_X', 'sequence_group_num'] = 1\ndf_train_seq.loc[df_train_seq.target_id == '4V65_A1', 'sequence_group_num'] = 1\ndf_train_seq.loc[df_train_seq.target_id == '4V65_BB', 'sequence_group_num'] = 1\ndf_train_seq.loc[df_train_seq.target_id == '4V5F_CA', 'sequence_group_num'] = 1\n\ndf_train = df_train.fillna(0)\n\ndf_train[\"sequence_id\"] = df_train[\"ID\"].apply(lambda x: \"_\".join(x.split(\"_\")[:-1]))\ndf_train[\"sequence_target\"] = df_train[\"ID\"].apply(lambda x: \"\".join(x.split(\"_\")[0]))\ndf_train[\"sequence_class\"] = df_train[\"ID\"].apply(lambda x: \"\".join(x.split(\"_\")[1])) #\"_\".join(    #df_train[\"ID\"][1].split(\"_\")[0:2]\n\n\ntrain = pd.merge(df_train, df_train_seq[['sequence','temporal_cutoff',\t'description',\t'all_sequences',\t'sequence_id',\t'sequence_num',\t'sequence_group',\t'sequence_group_num']], on=\"sequence_id\")\n\n# ------------------\n# Valid data\n# ------------------\n# replace extreme values by NaN\ndf_valid.replace(to_replace=-1E18, value=np.nan, inplace=True);\n\ndf_valid_seq['sequence_id'] = df_valid_seq[\"target_id\"]\ndf_valid_seq[\"sequence_num\"] = df_valid_seq[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"|\")[0:1])[1:])\ndf_valid_seq[\"sequence_group\"] = df_valid_seq[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"_\")[0:1])[1:])\ndf_valid_seq[\"sequence_group_num\"] = df_valid_seq[\"sequence_num\"].apply(lambda x: \"_\".join(x.split(\"_\")[1:2]))\n\ndf_valid = df_valid.fillna(0)\n\ndf_valid[\"sequence_id\"] = df_valid[\"ID\"].apply(lambda x: \"_\".join(x.split(\"_\")[:-1]))\ndf_valid[\"sequence_target\"] = df_valid[\"ID\"].apply(lambda x: \"\".join(x.split(\"_\")[0]))\ndf_valid[\"sequence_class\"] = df_valid[\"ID\"].apply(lambda x: \"\".join(x.split(\"_\")[1])) #\"_\".join(    #df_train[\"ID\"][1].split(\"_\")[0:2]\n\nvalid = pd.merge(df_valid[['ID',\t'resname',\t'resid',\t'x_1',\t'y_1',\t'z_1', 'sequence_id',\t'sequence_target',\t'sequence_class']], \\\n                 df_valid_seq[['sequence','temporal_cutoff',\t'description',\t'all_sequences',\t'sequence_id',\t'sequence_num',\t'sequence_group',\t'sequence_group_num']], on=\"sequence_id\")\n\n# ------------------\n# Test data\n# ------------------\n\ndf_test['sequence_id'] = df_test[\"target_id\"]\ndf_test[\"sequence_num\"] = df_test[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"|\")[0:1])[1:])\ndf_test[\"sequence_group\"] = df_test[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"_\")[0:1])[1:])\ndf_test[\"sequence_group_num\"] = df_test[\"sequence_num\"].apply(lambda x: \"\".join(x.split(\"_\")[1:2]))\n\n\ndf_test_seq = df_test.copy()\n\ndf_test_seq['resname'] = df_test_seq['sequence'].apply(list)  # Преобразуем каждую строку в список букв\ndf_test_seq['target_id'+'sequence'] = df_test_seq['target_id']+df_test_seq['sequence']\n\n# create new dataset for every resid by row\ntest = df_test_seq.explode('resname')  \ntest['resid'] = test.groupby('target_idsequence').cumcount() + 1\ntest['x_1'] = 0\ntest['y_1'] = 0\ntest['z_1'] = 0\n\ntest[\"sequence_target\"] = test[\"target_id\"].apply(lambda x: \"\".join(x.split(\"_\")[0]))\ntest[\"sequence_class\"] = 1\ntest[\"ID\"] = test[\"target_id\"]+\"_\"+test[\"resid\"].astype(str)\n\ntest = test[['ID','resname'\t,'resid',\t'x_1',\t'y_1',\t'z_1',\t'sequence_id',\t'sequence_target',\t'sequence_class',\t'sequence',\t'temporal_cutoff',\t'description'\t,'all_sequences', 'sequence_num'\t,'sequence_group'\t,'sequence_group_num']]\n\ndel df_train ,df_valid ,df_train_seq ,df_valid_seq ,df_test \n\ntarget_col = ['x_1', 'y_1', 'z_1']\ntrain[target_col] = train[target_col].fillna(train[target_col].mean())\n\ndisplay(train[1:2])\ndisplay(valid[1:2])\ndisplay(test[1:2])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:22:49.216424Z","iopub.execute_input":"2025-12-01T10:22:49.216697Z","iopub.status.idle":"2025-12-01T10:22:49.564800Z","shell.execute_reply.started":"2025-12-01T10:22:49.216671Z","shell.execute_reply":"2025-12-01T10:22:49.564065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, TransformerMixin\nclass AggFeatureExtractor(BaseEstimator, TransformerMixin):\n    \n    def __init__(self, group_col, agg_col, agg_func):\n        self.group_col = group_col\n        self.group_col_name = ''\n        for col in group_col:\n            self.group_col_name += col\n        self.agg_col = agg_col\n        self.agg_func = agg_func\n        self.agg_df = None\n        self.medians = None\n        \n    def fit(self, X, y=None):\n        group_col = self.group_col\n        agg_col = self.agg_col\n        agg_func = self.agg_func\n        \n        self.agg_df = X.groupby(group_col)[agg_col].agg(agg_func)\n        self.agg_df.columns = [f'{self.group_col_name}_{agg}_{_agg_col}' for _agg_col in agg_col for agg in agg_func]\n        self.medians = X[agg_col].median()\n        \n        return self\n    \n    def transform(self, X):\n        group_col = self.group_col\n        agg_col = self.agg_col\n        agg_func = self.agg_func\n        agg_df = self.agg_df\n        medians = self.medians\n        \n        X_merged = pd.merge(X, agg_df, left_on=group_col, right_index=True, how='left')\n        X_merged.fillna(medians, inplace=True)\n        X_agg = X_merged.loc[:, [f'{self.group_col_name}_{agg}_{_agg_col}' for _agg_col in agg_col for agg in agg_func]]\n        \n        return X_agg\n    \n    def fit_transform(self, X, y=None):\n        self.fit(X, y)\n        X_agg = self.transform(X)\n        return X_agg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:22:49.566221Z","iopub.execute_input":"2025-12-01T10:22:49.566424Z","iopub.status.idle":"2025-12-01T10:22:49.572086Z","shell.execute_reply.started":"2025-12-01T10:22:49.566408Z","shell.execute_reply":"2025-12-01T10:22:49.571391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_atoms_in_region(x, y, z, x_target, y_target, z_target, tolerance=100):\n    \"\"\"\n    Filter all RNA atoms that are within a given tolerance range around a target coordinate.\n    \"\"\"\n    filtered_atoms =  (min(x_target - tolerance, x_target + tolerance) < x < max(x_target - tolerance, x_target + tolerance)) \\\n        & (min(y_target - tolerance, y_target + tolerance) < y < max(y_target - tolerance, y_target + tolerance)) \\\n        & (min(z_target - tolerance, z_target + tolerance) < z < max(z_target - tolerance, z_target + tolerance))\n\n    return filtered_atoms\n    \ndef euclidean_distance(x1,y1,z1,x2,y2,z2):\n    return np.sqrt((x2 - x1)**2 +\n                   (y2 - y1)**2 +\n                   (z2 - z1)**2)\n\ndef count_nucleotides_map(sequence):\n    return {\n        'A': sequence.count('A'),\n        'C': sequence.count('C'),\n        'G': sequence.count('G'),\n        'U': sequence.count('U')\n    }\ndef count_nucleotides(sequence):\n      return    f\"A{sequence.count('A')}C{sequence.count('C')}G{sequence.count('G')}U{sequence.count('U')}\"\n\ndef get_nucleotide_combinations(seq, start_pos, length):\n        if start_pos + length <= len(seq):\n            result = seq[start_pos:start_pos + length]\n        else:\n            result = '-'  \n        return result\n    \ndef FE (df):\n    df['temporal_cutoff'] = pd.to_datetime(df['temporal_cutoff']).astype('int64') // 10**9\n    df[\"res\"] = df[\"resname\"] + df[\"resid\"].astype(str)\n                                    \n    df[\"length\"] = df[\"sequence\"].str.len()\n    \n    df['A_cnt'] = df['sequence'].astype(str).str.count(\"A\")\n    df['C_cnt'] = df['sequence'].astype(str).str.count(\"C\")\n    df['U_cnt'] = df['sequence'].astype(str).str.count(\"U\")\n    df['G_cnt'] = df['sequence'].astype(str).str.count(\"G\")\n    df['AC_cnt'] = df['sequence'].astype(str).str.count(\"AC\")\n    df['AU_cnt'] = df['sequence'].astype(str).str.count(\"AU\")\n    df['AG_cnt'] = df['sequence'].astype(str).str.count(\"AG\")\n    df['CA_cnt'] = df['sequence'].astype(str).str.count(\"CA\")\n    df['CU_cnt'] = df['sequence'].astype(str).str.count(\"CU\")\n    df['CG_cnt'] = df['sequence'].astype(str).str.count(\"CG\")\n    df['UA_cnt'] = df['sequence'].astype(str).str.count(\"UA\")\n    df['UC_cnt'] = df['sequence'].astype(str).str.count(\"UC\")\n    df['UG_cnt'] = df['sequence'].astype(str).str.count(\"UG\")\n    df['GA_cnt'] = df['sequence'].astype(str).str.count(\"GA\")\n    df['GC_cnt'] = df['sequence'].astype(str).str.count(\"GC\")\n    df['GU_cnt'] = df['sequence'].astype(str).str.count(\"GU\")\n    df['AA_cnt'] = df['sequence'].astype(str).str.count(\"AA\")\n    df['CC_cnt'] = df['sequence'].astype(str).str.count(\"CC\")\n    df['UU_cnt'] = df['sequence'].astype(str).str.count(\"UU\")\n    df['GG_cnt'] = df['sequence'].astype(str).str.count(\"GG\")\n    \n    df['begin_seq'] = df['sequence'].astype(str).str[0]\n    df['end_seq'] = df['sequence'].astype(str).str[-1]\n\n    df['nucleotide_ngram2'] = df.apply(lambda row: get_nucleotide_combinations( row['sequence'], row['resid']-1, 2),   axis=1)\n    df['nucleotide_ngram3'] = df.apply(lambda row: get_nucleotide_combinations( row['sequence'], row['resid']-1, 3),   axis=1)\n    df['nucleotide_ngram4'] = df.apply(lambda row: get_nucleotide_combinations( row['sequence'], row['resid']-1, 4),   axis=1)\n    df['nucleotide_ngram2_prev'] = df['nucleotide_ngram2'].shift(1)\n    df['nucleotide_ngram2_prev'].replace(to_replace=np.nan, value=\"<\", inplace=True)\n  \n    #df['nucleotide_counts'] = df['sequence'].apply(count_nucleotides)\n    df['counts_a'] = df['sequence'].apply(lambda x: x.count('A'))\n    df['counts_c'] = df['sequence'].apply(lambda x: x.count('C'))\n    df['counts_g'] = df['sequence'].apply(lambda x: x.count('G'))\n    df['counts_u'] = df['sequence'].apply(lambda x: x.count('U'))\n\n    df[\"GC_content\"] = df[\"sequence\"].apply(\n        lambda seq: (seq.count(\"G\") + seq.count(\"C\")) / len(seq) )\n    df[\"GA_content\"] = df[\"sequence\"].apply(\n        lambda seq: (seq.count(\"G\") + seq.count(\"A\")) / len(seq) )\n    df[\"GU_content\"] = df[\"sequence\"].apply(\n        lambda seq: (seq.count(\"G\") + seq.count(\"U\")) / len(seq) )\n    df[\"CA_content\"] = df[\"sequence\"].apply(\n        lambda seq: (seq.count(\"C\") + seq.count(\"A\")) / len(seq) )\n    df[\"CU_content\"] = df[\"sequence\"].apply(\n        lambda seq: (seq.count(\"C\") + seq.count(\"U\")) / len(seq) )\n    df[\"AU_content\"] = df[\"sequence\"].apply(\n        lambda seq: (seq.count(\"A\") + seq.count(\"U\")) / len(seq) )\n\n    df['resname_prev'] = df['resname'].shift(1)\n    df['resname_next'] = df['resname'].shift(-1)\n    df['resname_prev'].replace(to_replace=np.nan, value=\"<\", inplace=True)\n    df['resname_next'].replace(to_replace=np.nan, value=\">\", inplace=True)\n\n    df['prev_x'] = df['x_1'].shift(1)\n    df['prev_y'] = df['y_1'].shift(1)\n    df['prev_z'] = df['z_1'].shift(1)\n    df['next_x'] = df['x_1'].shift(-1)\n    df['next_y'] = df['y_1'].shift(-1)\n    df['next_z'] = df['z_1'].shift(-1)\n\n    df[\"distance_from_origin\"] = np.sqrt(df[\"x_1\"]**2 + df[\"y_1\"]**2 + df[\"z_1\"]**2 )\n\n    df['distance_prev'] = df.apply(lambda row: euclidean_distance( row[\"x_1\"], row[\"y_1\"], row[\"x_1\"],row['prev_x'], row['prev_y'], row['prev_z']), axis=1)\n    df['distance_next'] = df.apply(lambda row: euclidean_distance( row[\"x_1\"], row[\"y_1\"], row[\"x_1\"],row['next_x'], row['next_y'], row['next_z']), axis=1)\n\n    df[\"chain\"] = df[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"|\")[1:2])[6:].replace('auth', '').replace(' ', '').strip())\n    df[\"RNA\"] = df[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"|\")[2:3])[0:])  \n    df[\"RNA_chain\"] = df[\"RNA\"].astype(str).apply(lambda x: \"_\".join(x.split(\"(5'-R(\")[1:])[0:].replace(\")-3')\",\"\").replace(\") -3')\",\"\"))  \n    df[\"add_chain\"] = df[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"|\")[3:][4:5] )[6:].replace('auth', '').replace(' ', '').strip())\n    df[\"add_RNA\"] = df[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"|\")[3:][5:6] ))\n    df[\"type_RNA\"] = df[\"all_sequences\"].astype(str).apply(lambda x: \"_\".join(x.split(\"|\")[3:][6:7] ))  \n    df[\"class\"] = df[\"type_RNA\"].astype(str).apply(lambda x: \"_\".join(x.split(\"\\n\")[0:1] ))\n    df[\"class_chain\"] = df[\"type_RNA\"].astype(str).apply(lambda x: \"_\".join(x.split(\"\\n\")[1:] ))\n    \n    df = df.fillna(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:22:49.573036Z","iopub.execute_input":"2025-12-01T10:22:49.573208Z","iopub.status.idle":"2025-12-01T10:22:49.593125Z","shell.execute_reply.started":"2025-12-01T10:22:49.573193Z","shell.execute_reply":"2025-12-01T10:22:49.591993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------\n# Prepare dataset for model \n# ------------------\nprint(f\"Prepare datasets for model \")\n\ntrain['set'] = 'train'\nvalid['set'] = 'valid'\ntest['set'] = 'test'\n\nX_combine = pd.concat([train,valid])\nX_combine = pd.concat([X_combine,test])\n\n# ------------------\n# Create Features\n# ------------------\n\nFE(X_combine)\n\n\nFEATURES = list(set(X_combine.columns.tolist()) - set(train.columns.tolist()))\nprint(f\"create :{len(X_combine.columns.tolist()) - len(train.columns.tolist())} features,\\n \", f\"for columns :{FEATURES} \\n \")\n\nprint(f\"train + valid + test shape :{blu}{X_combine.shape}{clr}, \")\nprint(f'{\"-\" * 100}')\n# ------------------\n# Aggregate Features\n# ------------------\n#group_cols = [\n#        ['sequence_group'],['resname'],['res'],['nucleotide_ngram2'],\t['nucleotide_ngram3'],\t['nucleotide_ngram4'], [\"chain\"]\t  #['res','nres']\n#    ]\n#agg_col = [\n#        'x_1',\n#        'y_1', \n#        'z_1'\n#    ]\ngroup_cols = [   #resname_prev\tresname_next\n        ['sequence_group'],['resname'],['res'],['nucleotide_ngram2'],['nucleotide_ngram2_prev'],\t[\"chain\",'res']\n    ]\nagg_col = [\n        'x_1','y_1', 'z_1',\n        #'prev_x','prev_y','prev_z','next_x','next_y','next_z','distance_prev','distance_next',\n        'distance_from_origin',\n    ]\n\n\ndef agg_train_df(df, group_cols,agg_col ):\n    \n    agg_train = []\n    print(f\"agregate feature \\n group by :{group_cols},\\n \", f\"for columns :{agg_col} \\n \")\n    for group_col in group_cols:\n            agg_extractor = AggFeatureExtractor(group_col=group_col, agg_col=agg_col, agg_func=['mean',  'min', 'max', 'std'])\n            agg_extractor.fit(df)\n            agg_train.append(agg_extractor.transform(X_combine))\n    df = pd.concat([df] + agg_train, axis=1).fillna(0)\n    return df\n\nX_combine = agg_train_df(X_combine,group_cols,agg_col)\n\ngroup_cols = [   #resname_prev\tresname_next\n    \t['nucleotide_ngram3'],\t['nucleotide_ngram4'], \n        ['resname_prev','res',\t'resname_next']\t  ,['sequence_group','res'],\n    ]\nagg_col = [\n        'x_1','y_1', 'z_1',\n        #'prev_x','prev_y','prev_z','next_x','next_y','next_z',\n        'distance_prev','distance_next','distance_from_origin',\n    ]\n\nX_combine = agg_train_df(X_combine,group_cols,agg_col)\n\nX_combine['x_window'] = X_combine.apply(lambda row: (row['sequence_group_max_x_1'] - row['sequence_group_min_x_1']),   axis=1)\nX_combine['y_window'] = X_combine.apply(lambda row: (row['sequence_group_max_y_1'] - row['sequence_group_min_y_1']),   axis=1)\nX_combine['z_window'] = X_combine.apply(lambda row: (row['sequence_group_max_z_1'] - row['sequence_group_min_z_1']),   axis=1)\n\ntolerance = 50\nX_combine[f'atom_round{tolerance}'] = X_combine.apply(lambda row: find_atoms_in_region( row[\"x_1\"], row[\"y_1\"],row[\"z_1\"],row['sequence_group_max_x_1'] - row['sequence_group_min_x_1'],row['sequence_group_max_y_1'] - row['sequence_group_min_y_1'],row['sequence_group_max_z_1'] - row['sequence_group_min_z_1'], tolerance),   axis=1)\ntolerance = 100\nX_combine[f'atom_round{tolerance}'] = X_combine.apply(lambda row: find_atoms_in_region( row[\"x_1\"], row[\"y_1\"],row[\"z_1\"],row['sequence_group_max_x_1'] - row['sequence_group_min_x_1'],row['sequence_group_max_y_1'] - row['sequence_group_min_y_1'],row['sequence_group_max_z_1'] - row['sequence_group_min_z_1'], tolerance),   axis=1)\ntolerance = 150\nX_combine[f'atom_round{tolerance}'] = X_combine.apply(lambda row: find_atoms_in_region( row[\"x_1\"], row[\"y_1\"],row[\"z_1\"],row['sequence_group_max_x_1'] - row['sequence_group_min_x_1'],row['sequence_group_max_y_1'] - row['sequence_group_min_y_1'],row['sequence_group_max_z_1'] - row['sequence_group_min_z_1'], tolerance),   axis=1)\ntolerance = 300\nX_combine[f'atom_round{tolerance}'] = X_combine.apply(lambda row: find_atoms_in_region( row[\"x_1\"], row[\"y_1\"],row[\"z_1\"],row['sequence_group_max_x_1'] - row['sequence_group_min_x_1'],row['sequence_group_max_y_1'] - row['sequence_group_min_y_1'],row['sequence_group_max_z_1'] - row['sequence_group_min_z_1'], tolerance),   axis=1)\n\n\none_hot_encoded = pd.get_dummies(X_combine['resname'], prefix='nucleotide')\n\nX_combine = pd.concat([X_combine, one_hot_encoded], axis=1)\n\nprint(f\"train + valid + test shape :{blu}{X_combine.shape}{clr}, \")\nprint(f'{\"-\" * 100}')\n\nX_combine.head(3)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:22:49.594170Z","iopub.execute_input":"2025-12-01T10:22:49.594474Z","iopub.status.idle":"2025-12-01T10:23:45.394588Z","shell.execute_reply.started":"2025-12-01T10:22:49.594447Z","shell.execute_reply":"2025-12-01T10:23:45.393471Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"For a text description, we will form a bag of words, then reduce the dimension of the new set of features","metadata":{}},{"cell_type":"code","source":"### ------------------\n# Create Features with bag word for Descripction and RNA columns\n# ------------------\ncount_vectorizer = CountVectorizer(\n    analyzer=\"word\", tokenizer=nltk.word_tokenize,\n    preprocessor=None, stop_words='english', max_features=None)    \nbag_of_words_combine = count_vectorizer.fit_transform(X_combine['description'])\nbag_of_words_combine1 = count_vectorizer.fit_transform(X_combine['RNA'])\n\nsvd = TruncatedSVD(n_components=20, n_iter=30, random_state=12)\ntruncated_bag_of_words_combine = svd.fit_transform(bag_of_words_combine)\nsvd1 = TruncatedSVD(n_components=5, n_iter=25, random_state=12)\ntruncated_bag_of_words_combine1 = svd1.fit_transform(bag_of_words_combine1)\n\nadd_col = [f'desc_{i}' for i in svd.get_feature_names_out()]\nadd_col1 = [f'rna_{i}' for i in svd1.get_feature_names_out()]\n\nbag_df = pd.DataFrame(truncated_bag_of_words_combine, columns = add_col )\nbag_df1 = pd.DataFrame(truncated_bag_of_words_combine1, columns = add_col1)\n\nX_combine = X_combine.reset_index(drop=True)\nX_combine = pd.concat([X_combine, bag_df1], axis = 1)\nprint(f\"train + valid + test shape :{blu}{X_combine.shape}{clr}, \")\n\nX_combine = X_combine.reset_index(drop=True)\nX_combine = pd.concat([X_combine, bag_df], axis = 1)\nprint(f\"train + valid + test shape :{blu}{X_combine.shape}{clr}, \")\n\ndel bag_df,bag_df1\n#X_test[add_col] = pd.DataFrame(truncated_bag_of_words_test, columns = add_col )\nX_combine.head(5)","metadata":{"execution":{"iopub.status.busy":"2025-12-01T10:23:45.395701Z","iopub.execute_input":"2025-12-01T10:23:45.396084Z","iopub.status.idle":"2025-12-01T10:24:10.512789Z","shell.execute_reply.started":"2025-12-01T10:23:45.396049Z","shell.execute_reply":"2025-12-01T10:24:10.511887Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------\n# Label encode features \n# ------------------\nRMV = ['x_1',\t'y_1',\t'z_1','prev_x',\t'prev_y',\t'prev_z',\t'next_x',\t'next_y',\t'next_z',\n       'distance_from_origin',\t'distance_prev',\t'distance_next', 'sequence_group_num'\t]    \n# from the final dataset we remove all columns calculated based on coordinates and leave those calculated by groups\n\nFEATURES = [c for c in X_combine.columns if not c in RMV]\n\nfor_delete = ['ID','all_sequences', 'set', 'description',\"type_RNA\",'sequence_id',\t'sequence_target'] #todo  sequence_id\tsequence_target\n\ncat_features = X_combine.drop(RMV, axis=1).select_dtypes(include=['object']).columns.tolist()\nnum_features = X_combine.drop(RMV, axis=1).select_dtypes(exclude=['object']).columns.tolist()\n\nfor_label_encode = [c for c in cat_features if not c in for_delete] \n\nprint(f\"Category features will be encoded: \",for_label_encode)\nprint(f\"\\nNumerical: \",num_features)\n\nfor i,c in enumerate(for_label_encode):\n    combine = X_combine[c]\n    combine,_ = pd.factorize(combine)\n    X_combine[c] = combine.astype(\"float32\")  # int32\n  \n\n\n# ------------------\n# Result datasets for model \n# ------------------\nX_train = X_combine[X_combine['set'] == 'train']\nX_valid = X_combine[X_combine['set'] == 'valid']\nX_test = X_combine[X_combine['set'] == 'test']\n\nX_combine = X_combine.drop(for_delete, axis =1, errors='ignore')\nX_train = X_train.drop(for_delete, axis =1, errors='ignore')\nX_valid = X_valid.drop(for_delete, axis =1, errors='ignore')\nX_test = X_test.drop(for_delete, axis =1, errors='ignore')\n\nFEATURES = [c for c in X_train.columns if not c in RMV]\ntarget_col = ['x_1', 'y_1', 'z_1']\n\nprint(f\"train shape :{blu}{X_train.shape}{clr}, \", f\"valid shape :{blu}{X_valid.shape}{clr}, \", f\"test shape :{blu}{X_test.shape}{clr}\")\n\nprint(f\"X_train ->  isnull :{X_train.isnull().values.sum()}\")\nprint(f\"X_valid ->  isnull :{X_valid.isnull().values.sum()}\")\nprint(f\"X_test -> isnull :{X_test.isnull().values.sum()}\")\n\nX_train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:10.513618Z","iopub.execute_input":"2025-12-01T10:24:10.513860Z","iopub.status.idle":"2025-12-01T10:24:11.542364Z","shell.execute_reply.started":"2025-12-01T10:24:10.513837Z","shell.execute_reply":"2025-12-01T10:24:11.541066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df\n\n\n\nreduce_mem_usage(X_combine)\nreduce_mem_usage(X_train)\nreduce_mem_usage(X_valid)\nreduce_mem_usage(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:11.544764Z","iopub.execute_input":"2025-12-01T10:24:11.545072Z","iopub.status.idle":"2025-12-01T10:24:12.309760Z","shell.execute_reply.started":"2025-12-01T10:24:11.545049Z","shell.execute_reply":"2025-12-01T10:24:12.309038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FEATURES = [\n    \"atom_round150\",\n    \"atom_round50\",\n    \"CG_cnt\",\n    \"chain\",\n    \"chainres_max_x_1\",\n    \"chainres_max_y_1\",\n    \"chainres_max_z_1\",\n    \"CC_cnt\",\n    \"AU_cnt\",\n    #\"chainres_min_x_1\",\n    \"desc_truncatedsvd1\",\n    \"desc_truncatedsvd11\",\n    #\"desc_truncatedsvd12\",\n    \"desc_truncatedsvd16\",\n    #\"desc_truncatedsvd8\",\n    #\"desc_truncatedsvd9\",\n    \"nucleotide_ngram3_min_x_1\",\n    \"nucleotide_ngram3_min_y_1\",\n    \"nucleotide_ngram3_min_z_1\",\n    \"nucleotide_ngram4_mean_distance_from_origin\",\n    \"res_max_distance_from_origin\",\n    \"res_std_distance_from_origin\",\n    \"resname_mean_x_1\",\n    \"resname_mean_y_1\",\n    \"resname_mean_z_1\",\n    \"resname_prevresresname_next_max_x_1\",\n    \"resname_prevresresname_next_max_y_1\",\n    \"resname_prevresresname_next_max_z_1\",\n    \"nucleotide_ngram3_mean_distance_from_origin\",\n    \"resname_prevresresname_next_mean_x_1\",\n    \"resname_prevresresname_next_mean_y_1\",\n    \"resname_prevresresname_next_mean_z_1\",\n    \"RNA\",\n    \"rna_truncatedsvd0\",\n    \"resname_prevresresname_next_min_x_1\",\n    \"resname_prevresresname_next_min_y_1\",\n    \"resname_prevresresname_next_min_z_1\",\n    \"sequence_groupres_mean_x_1\",\n    \"sequence_groupres_mean_y_1\",\n    \"sequence_groupres_mean_z_1\",\n    \"sequence_group_mean_x_1\",\n    \"sequence_group_mean_y_1\",\n    \"sequence_group_mean_z_1\",\n    \"sequence_group_min_x_1\",\n    \"sequence_group_min_y_1\",\n    \"sequence_group_min_z_1\",\n    \"chainres_min_x_1\",\n    \"chainres_min_y_1\",\n    \"chainres_min_z_1\",\n    'x_window','y_window','z_window'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:12.310863Z","iopub.execute_input":"2025-12-01T10:24:12.311022Z","iopub.status.idle":"2025-12-01T10:24:12.314912Z","shell.execute_reply.started":"2025-12-01T10:24:12.311008Z","shell.execute_reply":"2025-12-01T10:24:12.314312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if 0:\n    corr_matrix = X_train[target_col+FEATURES].corr().abs()\n\n    threshold = 0.3\n    \n    filtered_corr_df = corr_matrix[(corr_matrix >= threshold) & (corr_matrix != 1.000)] \n    \n    plt.figure(figsize=(30,30))\n    sns.heatmap(filtered_corr_df, annot=True, cmap=\"Reds\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:12.315488Z","iopub.execute_input":"2025-12-01T10:24:12.315660Z","iopub.status.idle":"2025-12-01T10:24:12.334065Z","shell.execute_reply.started":"2025-12-01T10:24:12.315639Z","shell.execute_reply":"2025-12-01T10:24:12.333047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#FEATURES = ['resname','resid'] + list_f + add_col + add_col1\nFEATURES = [c for c in X_combine.columns if not c in RMV]\nprint(len(FEATURES))\n#FEATURES","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:12.334914Z","iopub.execute_input":"2025-12-01T10:24:12.335141Z","iopub.status.idle":"2025-12-01T10:24:12.347607Z","shell.execute_reply.started":"2025-12-01T10:24:12.335121Z","shell.execute_reply":"2025-12-01T10:24:12.346676Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  Data Modeling","metadata":{}},{"cell_type":"code","source":"def calculate_tm_score_exact(pred_coords, true_coords):\n    \"\"\"\n    Implementation closer to the official method used by US-align.\n    \"\"\"\n    # Remove padding\n    mask = ~np.all(true_coords == 0)\n    pred = pred_coords[mask]\n    true = true_coords[mask]\n    \n    Lref = len(true_coords)\n    \n    # Define d0 exactly as in the evaluation formula\n    if Lref >= 30:\n        d0 = 0.6 * np.sqrt(Lref - 0.5) - 2.5\n    elif Lref >= 24:\n        d0 = 0.7\n    elif Lref >= 20:\n        d0 = 0.6\n    elif Lref >= 16:\n        d0 = 0.5\n    elif Lref >= 12:\n        d0 = 0.4\n    else:\n        d0 = 0.3\n    \n    # Normalize structures\n    pred_centered = pred - np.mean(pred, axis=0)\n    true_centered = true - np.mean(true, axis=0)\n    \n    # Covariance matrix for optimal rotation\n    covariance = np.dot(pred_centered.T, true_centered)\n    U, S, Vt = np.linalg.svd(covariance)\n    rotation = np.dot(U, Vt)\n    \n    # Apply rotation\n    pred_aligned = np.dot(pred_centered, rotation)\n    \n    # Calculate distances\n    distances = np.sqrt(np.sum((pred_aligned - true_centered) ** 2, axis=1))\n    \n    # Calculate TM-score terms\n    tm_terms = 1.0 / (1.0 + (distances / d0) ** 2)\n    tm_score = np.sum(tm_terms) / Lref\n    \n    return float(tm_score)\n    \ndef calculate_tm_score(pred_coords, true_coords, d0_scale=1.24):\n    \"\"\"\n    Calculates a robust approximation of the TM-score between predicted and true coordinates.\n    Adds protection against division by zero and NaN.\n    \"\"\"\n    # Remove padding (rows with zeros) from true structures\n    mask = ~np.all(true_coords == 0)\n    pred = pred_coords[mask]\n    true = true_coords[mask]\n    \n    L = len(true_coords)\n\n    if L < 3:\n        return 0.0\n    \n    # Define d0 based on L (values adapted for RNA)\n    if L >= 30:\n        d0 = 0.6 * np.sqrt(L - 0.5) - 2.5\n        d0 = max(0.1, d0)\n    elif L >= 24:\n        d0 = 0.7\n    elif L >= 20:\n        d0 = 0.6\n    elif L >= 16:\n        d0 = 0.5\n    elif L >= 12:\n        d0 = 0.4\n    else:\n        d0 = 0.3\n    \n    distances = np.sqrt(np.sum((pred - true) ** 2, axis=1))\n    tm_terms = 1.0 / (1.0 + (distances / (d0 + 1e-8)) ** 2)\n    tm_score = np.sum(tm_terms) / L\n    return float(tm_score)\n\ndef plot_projection(pred):\n# Plot the results\n    plt.figure(figsize=(15, 5))\n    s = 100\n    a = 0.4\n    \n    \n    plt.subplot(1, 6, 1)  \n    plt.scatter(valid['x_1'],valid['y_1'], edgecolor=\"k\",c=\"cornflowerblue\", s=s,alpha=a)\n    plt.title('X Y projection valid')\n    \n    plt.subplot(1, 6, 2)  \n    plt.scatter(valid['y_1'],valid['z_1'], edgecolor=\"k\",c=\"cornflowerblue\", s=s,alpha=a)\n    plt.title('Y Z projection valid')\n    \n    plt.subplot(1, 6, 3) \n    plt.scatter(valid['x_1'], valid['z_1'], edgecolor=\"k\",c=\"cornflowerblue\", s=s,alpha=a)\n    plt.title('X Z projection valid')\n    \n    plt.subplot(1, 6, 4)  \n    plt.scatter(pred[:, 0],pred[:, 1], edgecolor=\"k\",c=\"cornflowerblue\", s=s,alpha=a)\n    plt.title('X Y projection')\n    \n    plt.subplot(1, 6, 5)  \n    plt.scatter(pred[:, 1],pred[:, 2], edgecolor=\"k\",c=\"cornflowerblue\", s=s,alpha=a)\n    plt.title('Y Z projection')\n    \n    plt.subplot(1, 6, 6) \n    plt.scatter(pred[:, 0],pred[:, 1], edgecolor=\"k\",c=\"cornflowerblue\", s=s,alpha=a)\n    plt.title('X Z projection')\n    \n    plt.tight_layout()  \n    \n    plt.legend()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:12.348441Z","iopub.execute_input":"2025-12-01T10:24:12.348705Z","iopub.status.idle":"2025-12-01T10:24:12.363670Z","shell.execute_reply.started":"2025-12-01T10:24:12.348679Z","shell.execute_reply":"2025-12-01T10:24:12.362793Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# GradientBoostingRegressor","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold,cross_val_score,RepeatedKFold,GridSearchCV\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.multioutput import RegressorChain\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import confusion_matrix, classification_report,  auc, roc_curve\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.svm import SVC\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:12.364353Z","iopub.execute_input":"2025-12-01T10:24:12.364530Z","iopub.status.idle":"2025-12-01T10:24:12.561996Z","shell.execute_reply.started":"2025-12-01T10:24:12.364513Z","shell.execute_reply":"2025-12-01T10:24:12.560707Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MultiTaskElasticNet","metadata":{}},{"cell_type":"code","source":"from sklearn import linear_model\nfrom sklearn.ensemble import ExtraTreesRegressor,RandomForestRegressor\nfrom sklearn.linear_model import LinearRegression, RidgeCV\nfrom sklearn.neighbors import KNeighborsRegressor,RadiusNeighborsRegressor\nfrom sklearn.gaussian_process import GaussianProcessRegressor\nfrom sklearn.kernel_ridge import KernelRidge\nfrom sklearn.cross_decomposition import PLSCanonical, PLSRegression\nfrom sklearn.tree import DecisionTreeRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:12.562910Z","iopub.execute_input":"2025-12-01T10:24:12.563198Z","iopub.status.idle":"2025-12-01T10:24:12.599345Z","shell.execute_reply.started":"2025-12-01T10:24:12.563143Z","shell.execute_reply":"2025-12-01T10:24:12.598189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"reg = linear_model.MultiTaskElasticNetCV(cv=3, l1_ratio=0.55, max_iter=2000, tol=0.001, random_state=0, selection='cyclic')\nreg = linear_model.MultiTaskElasticNet( l1_ratio=0.95, max_iter=500, tol=0.001, random_state=0, selection='cyclic')\n#reg = linear_model.MultiTaskElasticNet(alpha=0.91, l1_ratio=0.95, max_iter=1000, tol=0.001, random_state=0, selection='cyclic')\n#reg = RadiusNeighborsRegressor(radius=1000.0)\nreg.fit(X_train[FEATURES], X_train[target_col],)\n\nprint(\"Model score: \",reg.score(X_valid[FEATURES], X_valid[target_col]))\noof = reg.predict(X_valid[FEATURES])\nprint(\"Valid shape: \", oof.shape)\npred = reg.predict(X_test[FEATURES])\nprint(\"Predictions shape: \",pred.shape)\ndisplay(pred)\nplot_projection(pred) \ny_valid = X_valid[target_col].copy()\ny_valid = y_valid.reset_index(drop = True)\n\ny_pred = pd.DataFrame(pred[:,0:3], columns = target_col)\n\ny_valid = np.nan_to_num(X_valid[target_col], nan=0.0)\ny_pred = np.nan_to_num(y_pred, nan=0.0)\ntm_scores = []\ntm_scores1 = []\n\nfor i in range(len(X_valid[target_col])):\n       \n        tm = calculate_tm_score(y_pred[i], y_valid[i])\n        tm1 = calculate_tm_score_exact(y_pred[i], y_valid[i])\n        tm_scores.append(tm)\n        tm_scores1.append(tm1)\navg_tm_score = np.mean(tm_scores)\nprint(f\"Approximate average TM-score: {avg_tm_score:.8f}\")\navg_tm_score1 = np.mean(tm_scores1)\nprint(f\"Approximate average TM-score exact: {avg_tm_score1:.8f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:12.600119Z","iopub.execute_input":"2025-12-01T10:24:12.600442Z","iopub.status.idle":"2025-12-01T10:24:47.156467Z","shell.execute_reply.started":"2025-12-01T10:24:12.600416Z","shell.execute_reply":"2025-12-01T10:24:47.155489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_selection import SequentialFeatureSelector\n\nif 0:\n    selector = SequentialFeatureSelector(reg ,n_features_to_select=30, direction='forward', scoring=\"neg_root_mean_squared_error\", cv=2)\n    \n    selector.fit_transform(X_train[FEATURES], X_train[target_col])\n    mask = selector.get_support() #list of booleans\n    new_features = [] # The list of your K best features\n    feature_names = FEATURES\n    for bool_val, feature in zip(mask, feature_names):\n        if bool_val:\n            new_features.append(feature)\n    FEATURES = new_features\n    #dataframe = pd.DataFrame(X_train, columns=new_features)\n    #dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:47.157500Z","iopub.execute_input":"2025-12-01T10:24:47.157891Z","iopub.status.idle":"2025-12-01T10:24:47.187500Z","shell.execute_reply.started":"2025-12-01T10:24:47.157870Z","shell.execute_reply":"2025-12-01T10:24:47.186328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#dataframe[list(set(new_features)-set(target_col))#","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:24:47.188576Z","iopub.execute_input":"2025-12-01T10:24:47.189178Z","iopub.status.idle":"2025-12-01T10:24:47.193224Z","shell.execute_reply.started":"2025-12-01T10:24:47.189115Z","shell.execute_reply":"2025-12-01T10:24:47.192271Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Essamble","metadata":{}},{"cell_type":"code","source":"from sklearn import linear_model\n# Fit estimators\nESTIMATORS = {\n    #+\"KNeighborsRegressor\": KNeighborsRegressor(n_neighbors=15, weights='distance', algorithm='kd_tree', leaf_size=30, p=2, metric='minkowski'),\n    #! \"RadiusNeighborsRegressor\": RadiusNeighborsRegressor(radius=100.0,weights='distance' , algorithm='kd_tree', leaf_size=30, p=1,),\n    # \"LinearRegression\": LinearRegression(),\n    # \"RidgeCV\": RidgeCV(alphas=(0.01, 5.0, 10.0), cv=3),\n    # \"GaussianProcessRegressor\": GaussianProcessRegressor(),?\n    #\"KernelRidge\": KernelRidge(),?\n    #\"Lars\": linear_model.Lars(),\n    #\"Lasso\": linear_model.Lasso(),\n    #\"LassoLars\": linear_model.LassoLars(),\n    #\"MultiTaskLassoCV\": linear_model.MultiTaskLassoCV(), cv\n    #+\"OrthogonalMatchingPursuit\": linear_model.OrthogonalMatchingPursuit(),\n    #\"RANSACRegressor1\": linear_model.RANSACRegressor(estimator=KNeighborsRegressor(n_neighbors=15, weights='distance', algorithm='kd_tree', leaf_size=30, p=2, metric='minkowski'),min_samples=500,random_state=0),\n# test ortogonal first    \"RANSACRegressor2\": linear_model.RANSACRegressor(estimator=linear_model.MultiTaskElasticNet( l1_ratio=0.95, max_iter=500, tol=0.001, random_state=0, selection='cyclic'),min_samples=500,random_state=0),\n    \"RANSACRegressor3\": linear_model.RANSACRegressor(estimator=linear_model.OrthogonalMatchingPursuit(tol = 0.00001), max_trials=1000,  loss='squared_error',   min_samples=2000,random_state=0),   # add clf\n    #\"PLSCanonical\": PLSCanonical(n_components=3,scale=True, algorithm='svd', max_iter=2000,), #svd\n    #\"PLSRegression\": PLSRegression(n_components=3,scale=False,  max_iter=300,), \n    #\"DecisionTreeRegressor\":DecisionTreeRegressor(max_depth=8),\n    #\"RandomForestRegressor\": RandomForestRegressor(n_estimators=100, max_features=10, random_state=0, criterion='absolute_error'),\n    #\"ExtraTreesRegressor\": ExtraTreesRegressor(n_estimators=100, max_features=10, random_state=0, criterion='absolute_error'), #squared_error”, “absolute_error”, “friedman_mse”, “poisson”\n    #\"MultiTaskLassoCV\": linear_model.MultiTaskLassoCV(cv=3),\n}\n\ny_test_predict = dict()\noof_test = dict()\nfor name, estimator in ESTIMATORS.items():\n    \n    print(f\"Model {name}\")\n    estimator.fit(X_train[FEATURES], X_train[target_col])\n\n    print(f\".... score: \",estimator.score(X_valid[FEATURES], X_valid[target_col]))\n    oof_test[name] = estimator.predict(X_valid[FEATURES])\n    y_test_predict[name] = estimator.predict(X_test[FEATURES])\n    #print(\"Valid shape: \", oof.shape)\n    #pred = estimator.predict(X_test[FEATURES])\n    #print(\"Predictions shape: \",y_test_predict[name].shape)\n    #display(pred)\n\n    #for j, est in enumerate(sorted(ESTIMATORS)):\n    \n    plot_projection(y_test_predict[name])    \n\n    y_valid = X_valid[target_col].copy()\n    y_valid = y_valid.reset_index(drop = True)\n    \n    y_pred = pd.DataFrame(y_test_predict[name][:,0:3], columns = target_col)\n    \n    y_valid = np.nan_to_num(y_valid, nan=0.0)\n    y_pred = np.nan_to_num(y_pred, nan=0.0)\n    \n    tm_scores = []\n    tm_scores1 = []\n    \n    for i in range(len(X_valid[target_col])):\n           \n            tm = calculate_tm_score(y_pred[i], y_valid[i])\n            tm1 = calculate_tm_score_exact(y_pred[i], y_valid[i])\n            tm_scores.append(tm)\n            tm_scores1.append(tm1)\n        \n    avg_tm_score = np.mean(tm_scores)\n    print(f\"Approximate average TM-score: {avg_tm_score:.8f}\")\n    avg_tm_score1 = np.mean(tm_scores1)\n    print(f\"Approximate average TM-score exact: {avg_tm_score1:.8f}\")\n    print(f'{\"-\" * 100}')","metadata":{"execution":{"iopub.status.busy":"2025-12-01T10:24:47.193982Z","iopub.execute_input":"2025-12-01T10:24:47.194211Z","iopub.status.idle":"2025-12-01T10:26:49.580371Z","shell.execute_reply.started":"2025-12-01T10:24:47.194187Z","shell.execute_reply":"2025-12-01T10:26:49.579339Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Make predict","metadata":{}},{"cell_type":"code","source":"pred=y_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:49.581426Z","iopub.execute_input":"2025-12-01T10:26:49.581808Z","iopub.status.idle":"2025-12-01T10:26:49.585313Z","shell.execute_reply.started":"2025-12-01T10:26:49.581787Z","shell.execute_reply":"2025-12-01T10:26:49.584429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(y_pred[:, 0].shape)\ny_pred[:, 0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:49.586128Z","iopub.execute_input":"2025-12-01T10:26:49.586367Z","iopub.status.idle":"2025-12-01T10:26:49.603036Z","shell.execute_reply.started":"2025-12-01T10:26:49.586350Z","shell.execute_reply":"2025-12-01T10:26:49.601838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(pred[:, 0].shape)\npred[:, 0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:49.604245Z","iopub.execute_input":"2025-12-01T10:26:49.604574Z","iopub.status.idle":"2025-12-01T10:26:49.618789Z","shell.execute_reply.started":"2025-12-01T10:26:49.604540Z","shell.execute_reply":"2025-12-01T10:26:49.617576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nsub = pd.read_csv('../input/stanford-rna-3d-folding/sample_submission.csv')\ncol_name = ['x_1',\t'y_1',\t'z_1',\t'x_2',\t'y_2',\t'z_2',\t'x_3',\t'y_3',\t'z_3',\t'x_4',\t'y_4',\t'z_4',\t'x_5',\t'y_5',\t'z_5']\nsub = sub.drop(col_name, axis =1)\nsub['x_1'] = pred[:, 0]\nsub['y_1'] = pred[:, 1]\nsub['z_1'] = pred[:, 2]\n\nsub['x_2'] = 0.0\nsub['y_2'] = 0.0\nsub['z_2'] = 0.0\n\nsub['x_3'] = 0.0\nsub['y_3'] = 0.0\nsub['z_3'] = 0.0\n\nsub['x_4'] = 0.0\nsub['y_4'] = 0.0\nsub['z_4'] = 0.0\n\nsub['x_5'] = 0.0\nsub['y_5'] = 0.0\nsub['z_5'] = 0.0\n\n#for i in range(1,6):\n #   columns+=[f\"x_{i}\"]\n#    columns+=[f\"y_{i}\"]\n #   columns+=[f\"z_{i}\"]\n\nVER = 1\nsub.to_csv('submission.csv', index=False)\nsub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:49.619600Z","iopub.execute_input":"2025-12-01T10:26:49.619907Z","iopub.status.idle":"2025-12-01T10:26:49.689928Z","shell.execute_reply.started":"2025-12-01T10:26:49.619885Z","shell.execute_reply":"2025-12-01T10:26:49.688696Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualize predictions","metadata":{}},{"cell_type":"code","source":"def plot_structure(df: pd.DataFrame, sequence_id: str) -> None:\n    sequence_df = df[df[\"sequence_id\"] == sequence_id]\n    sequence_points = sequence_df[[\"x_1\", \"y_1\", \"z_1\", \"resname\"]]\n    seq_lst = sequence_df['resname'].to_list()\n    seq_str = ''.join(seq_lst)\n    print(seq_str)\n    #print(sequence_points)\n    \n    colors = {\"A\": \"red\", \"G\": \"blue\", \"C\": \"green\", \"U\": \"orange\"}\n    fig = go.Figure()\n    \n    for resname, color in colors.items():\n        subset = sequence_df[sequence_df[\"resname\"] == resname]\n        fig.add_trace(go.Scatter3d(\n            x=subset[\"x_1\"], y=subset[\"y_1\"], z=subset[\"z_1\"],\n            mode='markers',\n            marker=dict(size=5, color=color),\n            name=resname,\n            opacity=0.8\n        ))\n    \n    fig.add_trace(go.Scatter3d(\n        x=sequence_df[\"x_1\"], y=sequence_df[\"y_1\"], z=sequence_df[\"z_1\"],\n        mode='lines',\n        line=dict(color='gray', width=2),\n        name='RNA Backbone'\n    ))\n    \n    fig.update_layout(\n            scene=dict(xaxis_title='X', yaxis_title='Y', zaxis_title='Z'),\n            title=f'3D RNA Structure of sequence {sequence_id}',\n        )\n            \n    fig.show(renderer=\"iframe\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:49.692877Z","iopub.execute_input":"2025-12-01T10:26:49.693110Z","iopub.status.idle":"2025-12-01T10:26:49.699275Z","shell.execute_reply.started":"2025-12-01T10:26:49.693093Z","shell.execute_reply":"2025-12-01T10:26:49.697975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sequence_id = \"8UYS\"\nsequence_df = valid[valid[\"sequence_group\"] == sequence_id].copy()\n\nsequence_points = sequence_df[[\"x_1\", \"y_1\", \"z_1\", \"resname\"]]\nseq_lst = sequence_df['resname'].to_list()\nseq_str = ''.join(seq_lst)\nsequence_df[0:1]\nplot_structure(sequence_df, sequence_df[\"sequence_id\"].max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:49.700546Z","iopub.execute_input":"2025-12-01T10:26:49.700831Z","iopub.status.idle":"2025-12-01T10:26:50.228644Z","shell.execute_reply.started":"2025-12-01T10:26:49.700807Z","shell.execute_reply":"2025-12-01T10:26:50.227258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_id = sub[\"ID\"].apply(lambda x: \"_\".join(x.split(\"_\")[:-1])).unique()\n\ndef print_plot_str(c):\n    print(\"----------------------------------------------------------------\")\n    print(c)\n    sequence_df = sub[sub[\"sequence_group\"] == c].copy()\n    #display(sequence_df.head(3))\n    plot_structure(sequence_df, c)\n\nsub[\"sequence_group\"] = sub[\"ID\"].apply(lambda x: \"_\".join(x.split(\"_\")[:-1]))\nsub[\"sequence_id\"] = sub[\"ID\"].apply(lambda x: \"_\".join(x.split(\"_\")[:-1]))\nprint_plot_str('R1149')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.229676Z","iopub.execute_input":"2025-12-01T10:26:50.229973Z","iopub.status.idle":"2025-12-01T10:26:50.282094Z","shell.execute_reply.started":"2025-12-01T10:26:50.229943Z","shell.execute_reply":"2025-12-01T10:26:50.281002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1108')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.282793Z","iopub.execute_input":"2025-12-01T10:26:50.283079Z","iopub.status.idle":"2025-12-01T10:26:50.327399Z","shell.execute_reply.started":"2025-12-01T10:26:50.283052Z","shell.execute_reply":"2025-12-01T10:26:50.326684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1156')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.328941Z","iopub.execute_input":"2025-12-01T10:26:50.329398Z","iopub.status.idle":"2025-12-01T10:26:50.371010Z","shell.execute_reply.started":"2025-12-01T10:26:50.329363Z","shell.execute_reply":"2025-12-01T10:26:50.370276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1136')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.371717Z","iopub.execute_input":"2025-12-01T10:26:50.371932Z","iopub.status.idle":"2025-12-01T10:26:50.403180Z","shell.execute_reply.started":"2025-12-01T10:26:50.371912Z","shell.execute_reply":"2025-12-01T10:26:50.402404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1126')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.403857Z","iopub.execute_input":"2025-12-01T10:26:50.404086Z","iopub.status.idle":"2025-12-01T10:26:50.439401Z","shell.execute_reply.started":"2025-12-01T10:26:50.404065Z","shell.execute_reply":"2025-12-01T10:26:50.438351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1116')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.440192Z","iopub.execute_input":"2025-12-01T10:26:50.440440Z","iopub.status.idle":"2025-12-01T10:26:50.476108Z","shell.execute_reply.started":"2025-12-01T10:26:50.440419Z","shell.execute_reply":"2025-12-01T10:26:50.475051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1138')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.476894Z","iopub.execute_input":"2025-12-01T10:26:50.477182Z","iopub.status.idle":"2025-12-01T10:26:50.516705Z","shell.execute_reply.started":"2025-12-01T10:26:50.477145Z","shell.execute_reply":"2025-12-01T10:26:50.515755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1117v2')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.517776Z","iopub.execute_input":"2025-12-01T10:26:50.518053Z","iopub.status.idle":"2025-12-01T10:26:50.557565Z","shell.execute_reply.started":"2025-12-01T10:26:50.518034Z","shell.execute_reply":"2025-12-01T10:26:50.556809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1128')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.558283Z","iopub.execute_input":"2025-12-01T10:26:50.558531Z","iopub.status.idle":"2025-12-01T10:26:50.608031Z","shell.execute_reply.started":"2025-12-01T10:26:50.558509Z","shell.execute_reply":"2025-12-01T10:26:50.607297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1190')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.608756Z","iopub.execute_input":"2025-12-01T10:26:50.608963Z","iopub.status.idle":"2025-12-01T10:26:50.645848Z","shell.execute_reply.started":"2025-12-01T10:26:50.608946Z","shell.execute_reply":"2025-12-01T10:26:50.644712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1107')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.646656Z","iopub.execute_input":"2025-12-01T10:26:50.646886Z","iopub.status.idle":"2025-12-01T10:26:50.692801Z","shell.execute_reply.started":"2025-12-01T10:26:50.646855Z","shell.execute_reply":"2025-12-01T10:26:50.691873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_plot_str('R1189')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.693678Z","iopub.execute_input":"2025-12-01T10:26:50.693884Z","iopub.status.idle":"2025-12-01T10:26:50.743677Z","shell.execute_reply.started":"2025-12-01T10:26:50.693866Z","shell.execute_reply":"2025-12-01T10:26:50.742757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sequence_df = sub[sub[\"sequence_group\"] == 'R1138'].copy()\nX = sequence_df['x_1']\nY = sequence_df['y_1']\nZ = sequence_df['z_1']\n\n\nplt.figure(figsize=(15, 5))\n\nplt.subplot(1, 3, 1)  \nsns.scatterplot(data=sequence_df, x=X, y=Y, s=Z*Z.max()/X.max(),hue = 'resname')\nplt.title('X Y projection')\n\nplt.subplot(1, 3, 2)  \nsns.scatterplot(data=sequence_df, x=Y, y=Z, s=X*X.max()/Y.max(),hue = 'resname')\nplt.title('Y Z projection')\n\nplt.subplot(1, 3, 3) \nsns.scatterplot(data=sequence_df, x=X, y=Z, s=Y*Y.max()/X.max(),hue = 'resname')\nplt.title('X Z projection')\n\nplt.tight_layout()  \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:50.744285Z","iopub.execute_input":"2025-12-01T10:26:50.744464Z","iopub.status.idle":"2025-12-01T10:26:51.478478Z","shell.execute_reply.started":"2025-12-01T10:26:50.744447Z","shell.execute_reply":"2025-12-01T10:26:51.477508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"g = sns.pairplot(data=sequence_df[['x_1', 'y_1', 'z_1', 'resname']],\n             hue = 'resname', height=2.5,diag_kind=\"kde\",\n             diag_kws = {'color' : default_color_1},\n             plot_kws = {'s' : 15, \n                         'alpha' : 0.5,\n                         'color' : default_color_1})\ng.map_lower(sns.kdeplot, levels=1, color=\".2\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-01T10:26:51.479309Z","iopub.execute_input":"2025-12-01T10:26:51.479504Z","iopub.status.idle":"2025-12-01T10:26:55.461054Z","shell.execute_reply.started":"2025-12-01T10:26:51.479486Z","shell.execute_reply":"2025-12-01T10:26:55.460172Z"}},"outputs":[],"execution_count":null}]}