{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11228175,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Embedding, Conv1D, BatchNormalization, Dropout\nimport os\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.callbacks import EarlyStopping\nnp.random.seed(42)\ntf.random.set_seed(42)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:10:29.099078Z","iopub.execute_input":"2025-03-04T14:10:29.099451Z","iopub.status.idle":"2025-03-04T14:10:29.104820Z","shell.execute_reply.started":"2025-03-04T14:10:29.099421Z","shell.execute_reply":"2025-03-04T14:10:29.103813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_seq = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")\ntrain_lab = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_labels.csv\")\ntest_seq = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/test_sequences.csv\")\nval_lab = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/validation_labels.csv\")\nval_seq = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/validation_sequences.csv\")\nsub_sample = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/sample_submission.csv\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:10:32.416016Z","iopub.execute_input":"2025-03-04T14:10:32.416336Z","iopub.status.idle":"2025-03-04T14:10:32.687115Z","shell.execute_reply.started":"2025-03-04T14:10:32.416313Z","shell.execute_reply":"2025-03-04T14:10:32.686290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dfs = [train_seq, train_lab, test_seq, val_lab, val_seq, sub_sample]\n\nfor i, df in enumerate(dfs):\n    print(f\"DataFrame {i+1} Info:\")\n    print(df.info())\n    print(\"\\n\" + \"=\"*50 + \"\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:10:39.144597Z","iopub.execute_input":"2025-03-04T14:10:39.144923Z","iopub.status.idle":"2025-03-04T14:10:39.201234Z","shell.execute_reply.started":"2025-03-04T14:10:39.144897Z","shell.execute_reply":"2025-03-04T14:10:39.200437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_lab.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:10:47.221474Z","iopub.execute_input":"2025-03-04T14:10:47.221821Z","iopub.status.idle":"2025-03-04T14:10:47.244535Z","shell.execute_reply.started":"2025-03-04T14:10:47.221796Z","shell.execute_reply":"2025-03-04T14:10:47.243418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_lab.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:10:48.497397Z","iopub.execute_input":"2025-03-04T14:10:48.497794Z","iopub.status.idle":"2025-03-04T14:10:48.508414Z","shell.execute_reply.started":"2025-03-04T14:10:48.497763Z","shell.execute_reply":"2025-03-04T14:10:48.507589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i, df in enumerate(dfs):\n    print(f\"DataFrame {i+1} Info:\")\n    print(df.head())\n    print(\"\\n\" + \"=\"*50 + \"\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:10:49.813132Z","iopub.execute_input":"2025-03-04T14:10:49.813463Z","iopub.status.idle":"2025-03-04T14:10:49.846349Z","shell.execute_reply.started":"2025-03-04T14:10:49.813438Z","shell.execute_reply":"2025-03-04T14:10:49.845175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i, df in enumerate(dfs):\n    print(f\"DataFrame {i+1} Missing Values:\")\n    print(df.isnull().sum())  # Check for missing values\n    print(\"\\n\" + \"=\"*50 + \"\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:10:54.075141Z","iopub.execute_input":"2025-03-04T14:10:54.075556Z","iopub.status.idle":"2025-03-04T14:10:54.107843Z","shell.execute_reply.started":"2025-03-04T14:10:54.075523Z","shell.execute_reply":"2025-03-04T14:10:54.106855Z"}},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"train_seq[\"sequence_length\"] = train_seq[\"sequence\"].apply(len)\ntest_seq[\"sequence_length\"] = test_seq[\"sequence\"].apply(len)\n\n\nplt.figure(figsize=(10,5))\nsns.histplot(train_seq[\"sequence_length\"], bins=30, kde=True)\nplt.xlabel(\"RNA Sequence Length\")\nplt.ylabel(\"Count\")\nplt.title(\"Distribution of RNA Sequence Lengths in Training Data\")\nplt.show()\n","metadata":{}},{"cell_type":"code","source":"train_lab[['x_1', 'y_1', 'z_1']] = train_lab[['x_1', 'y_1', 'z_1']].fillna(train_lab[['x_1', 'y_1', 'z_1']].median())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:17:54.240822Z","iopub.execute_input":"2025-03-04T14:17:54.241138Z","iopub.status.idle":"2025-03-04T14:17:54.258558Z","shell.execute_reply.started":"2025-03-04T14:17:54.241116Z","shell.execute_reply":"2025-03-04T14:17:54.257554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i, df in enumerate(dfs):\n    print(f\"DataFrame {i+1} Missing Values:\")\n    print(df.isnull().sum())  # Check for missing values\n    print(\"\\n\" + \"=\"*50 + \"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:37:45.196797Z","iopub.execute_input":"2025-03-04T14:37:45.197099Z","iopub.status.idle":"2025-03-04T14:37:45.228871Z","shell.execute_reply.started":"2025-03-04T14:37:45.197077Z","shell.execute_reply":"2025-03-04T14:37:45.227773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_lab[['x_1', 'y_1', 'z_1']].hist(bins=30, figsize=(10, 5))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:38:58.778220Z","iopub.execute_input":"2025-03-04T14:38:58.778604Z","iopub.status.idle":"2025-03-04T14:38:59.587210Z","shell.execute_reply.started":"2025-03-04T14:38:58.778574Z","shell.execute_reply":"2025-03-04T14:38:59.586106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i, df in enumerate(dfs):\n    print(f\"DataFrame {i+1} Duplicates:\")\n    print(df.duplicated().sum())  \n    print(\"\\n\" + \"=\"*50 + \"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:51:59.297600Z","iopub.execute_input":"2025-03-04T14:51:59.297961Z","iopub.status.idle":"2025-03-04T14:51:59.378582Z","shell.execute_reply.started":"2025-03-04T14:51:59.297931Z","shell.execute_reply":"2025-03-04T14:51:59.377664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\n\ndef kmer_freq(sequence, k=3):\n    kmers = [sequence[i:i+k] for i in range(len(sequence) - k + 1)]\n    return Counter(kmers)\n\ntrain_seq['kmer_features'] = train_seq['sequence'].apply(lambda x: kmer_freq(x, k=3))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:53:12.452656Z","iopub.execute_input":"2025-03-04T14:53:12.452973Z","iopub.status.idle":"2025-03-04T14:53:12.460177Z","shell.execute_reply.started":"2025-03-04T14:53:12.452950Z","shell.execute_reply":"2025-03-04T14:53:12.459087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\n\nX = train_lab.drop(columns=['ID', 'resname', 'resid'])\ny = train_lab[['x_1', 'y_1', 'z_1']]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nmodel = RandomForestRegressor(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\npredictions = model.predict(X_test)\nprint(f'MSE: {mean_squared_error(y_test, predictions)}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T14:54:29.963540Z","iopub.execute_input":"2025-03-04T14:54:29.963904Z","iopub.status.idle":"2025-03-04T14:55:22.868256Z","shell.execute_reply.started":"2025-03-04T14:54:29.963872Z","shell.execute_reply":"2025-03-04T14:55:22.867224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv1D, Flatten, Dense, Dropout\n\ncnn_model = Sequential([\n    Conv1D(filters=64, kernel_size=3, activation='relu', input_shape=(X_train.shape[1], 1)),\n    Dropout(0.3),\n    Flatten(),\n    Dense(64, activation='relu'),\n    Dense(3)  # Predicting x, y, z coordinates\n])\n\ncnn_model.compile(optimizer='adam', loss='mse')\ncnn_model.fit(X_train, y_train, epochs=50, batch_size=32, validation_data=(X_test, y_test))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T16:06:55.402547Z","iopub.execute_input":"2025-03-04T16:06:55.402986Z","iopub.status.idle":"2025-03-04T16:11:18.726545Z","shell.execute_reply.started":"2025-03-04T16:06:55.402956Z","shell.execute_reply":"2025-03-04T16:11:18.725451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}