{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install biopython\n!pip install ViennaRNA","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T17:20:34.616347Z","iopub.execute_input":"2025-03-02T17:20:34.616820Z","iopub.status.idle":"2025-03-02T17:20:43.817409Z","shell.execute_reply.started":"2025-03-02T17:20:34.616760Z","shell.execute_reply":"2025-03-02T17:20:43.815735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom Bio.Seq import Seq\nfrom Bio.SeqUtils import  molecular_weight\nfrom Bio.SeqUtils import gc_fraction\nfrom Bio.SeqUtils import MeltingTemp as mt\nimport re\nimport RNA  ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-02T17:20:43.819392Z","iopub.execute_input":"2025-03-02T17:20:43.819797Z","iopub.status.idle":"2025-03-02T17:20:43.826145Z","shell.execute_reply.started":"2025-03-02T17:20:43.819749Z","shell.execute_reply":"2025-03-02T17:20:43.824501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv', usecols = ('target_id', 'sequence'))\ndf.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T17:20:43.828460Z","iopub.execute_input":"2025-03-02T17:20:43.828755Z","iopub.status.idle":"2025-03-02T17:20:43.882975Z","shell.execute_reply.started":"2025-03-02T17:20:43.828730Z","shell.execute_reply":"2025-03-02T17:20:43.881658Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Extracted Protein Sequence Features\n","metadata":{}},{"cell_type":"code","source":"def extract_rna_features(sequence):\n    sequence = re.sub(r'[^ACGU]', '', sequence.upper())\n    seq = Seq(sequence)\n    \n    # Basic statistics\n    length = len(sequence)\n    A_count = sequence.count('A')\n    C_count = sequence.count('C')\n    G_count = sequence.count('G')\n    U_count = sequence.count('U')\n    GC_content = gc_fraction(seq) * 100\n    \n    # Molecular weight\n    molecular_weight_value = molecular_weight(seq, seq_type='RNA')\n    \n    \n    # Purine/Pyrimidine Ratio\n    purines = A_count + G_count\n    pyrimidines = C_count + U_count\n    purine_pyrimidine_ratio = purines / pyrimidines if pyrimidines != 0 else None\n    \n    # Palindromic Motifs\n    palindromic_motifs = [sequence[i:j] for i in range(length) for j in range(i + 4, length + 1) if sequence[i:j] == sequence[i:j][::-1]]\n    palindromic_motifs_count = len(palindromic_motifs)\n    \n    # Start/Stop Codons (for coding RNA)\n    start_codon = sequence.startswith('AUG')\n    stop_codon = bool(re.search(r'(UAA|UGA|UAG)', sequence))\n    \n    return length, A_count, C_count, G_count, U_count, GC_content, molecular_weight_value, purine_pyrimidine_ratio, palindromic_motifs_count, start_codon, stop_codon","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T17:20:43.884487Z","iopub.execute_input":"2025-03-02T17:20:43.885031Z","iopub.status.idle":"2025-03-02T17:20:43.894294Z","shell.execute_reply.started":"2025-03-02T17:20:43.884989Z","shell.execute_reply":"2025-03-02T17:20:43.892720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[['length', 'A_count', 'C_count', 'G_count', 'U_count', \n    'GC_content', 'molecular_weight', \n    'purine_pyrimidine_ratio', 'palindromic_motifs_count', \n    'start_codon', 'stop_codon']] = df['sequence'].apply(lambda x: pd.Series(extract_rna_features(x)))\n\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T17:20:43.895942Z","iopub.execute_input":"2025-03-02T17:20:43.896403Z","iopub.status.idle":"2025-03-02T17:24:58.758378Z","shell.execute_reply.started":"2025-03-02T17:20:43.896349Z","shell.execute_reply":"2025-03-02T17:24:58.757090Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"here is link to df:\nhttps://www.kaggle.com/datasets/michalinahulak/stanford-rna-3d-folding-seq-feature-extraction","metadata":{}},{"cell_type":"code","source":"df.to_csv(\"protein_sequence_feature_etraction.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T17:24:58.760581Z","iopub.execute_input":"2025-03-02T17:24:58.760965Z","iopub.status.idle":"2025-03-02T17:24:58.786287Z","shell.execute_reply.started":"2025-03-02T17:24:58.760936Z","shell.execute_reply":"2025-03-02T17:24:58.784732Z"}},"outputs":[],"execution_count":null}]}