{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11403143,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:11:10.756939Z","iopub.execute_input":"2025-03-15T10:11:10.757329Z","iopub.status.idle":"2025-03-15T10:11:14.784322Z","shell.execute_reply.started":"2025-03-15T10:11:10.757288Z","shell.execute_reply":"2025-03-15T10:11:14.780898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# List available files in the competition directory\ncompetition_path = \"/kaggle/input/\"\nprint(\"Available datasets:\")\nprint(os.listdir(competition_path))\n\n# Check the folder name for this competition\ncompetition_folders = os.listdir(competition_path)\nfor folder in competition_folders:\n    print(f\"\\nFiles in {folder}:\")\n    print(os.listdir(os.path.join(competition_path, folder)))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:11:39.098870Z","iopub.execute_input":"2025-03-15T10:11:39.099291Z","iopub.status.idle":"2025-03-15T10:11:39.107724Z","shell.execute_reply.started":"2025-03-15T10:11:39.099256Z","shell.execute_reply":"2025-03-15T10:11:39.106402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Define the path\ndata_path = \"/kaggle/input/stanford-rna-3d-folding/\"\n\n# Load train data\ntrain_sequences = pd.read_csv(data_path + \"train_sequences.csv\")\ntrain_labels = pd.read_csv(data_path + \"train_labels.csv\")\n\n# Load test data\ntest_sequences = pd.read_csv(data_path + \"test_sequences.csv\")\n\n# Display first few rows\nprint(\"Train Sequences:\")\ndisplay(train_sequences.head())\n\nprint(\"\\nTrain Labels:\")\ndisplay(train_labels.head())\n\nprint(\"\\nTest Sequences:\")\ndisplay(test_sequences.head())\n\n# Check for missing values\nprint(\"\\nMissing Values in Train Sequences:\")\nprint(train_sequences.isnull().sum())\n\nprint(\"\\nMissing Values in Train Labels:\")\nprint(train_labels.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:12:06.674705Z","iopub.execute_input":"2025-03-15T10:12:06.675060Z","iopub.status.idle":"2025-03-15T10:12:07.239792Z","shell.execute_reply.started":"2025-03-15T10:12:06.675033Z","shell.execute_reply":"2025-03-15T10:12:07.238622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop rows with missing 3D coordinate values\ntrain_labels_clean = train_labels.dropna()\n\n# Drop rows with missing sequences\ntrain_sequences_clean = train_sequences.dropna()\n\nprint(f\"Remaining Rows in Train Labels: {len(train_labels_clean)}\")\nprint(f\"Remaining Rows in Train Sequences: {len(train_sequences_clean)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:12:58.436227Z","iopub.execute_input":"2025-03-15T10:12:58.436573Z","iopub.status.idle":"2025-03-15T10:12:58.480003Z","shell.execute_reply.started":"2025-03-15T10:12:58.436546Z","shell.execute_reply":"2025-03-15T10:12:58.478884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract target_id from ID column\ntrain_labels['target_id'] = train_labels['ID'].apply(lambda x: \"_\".join(x.split(\"_\")[:2]))\n\n# Merge datasets\ntrain_data = pd.merge(train_sequences, train_labels, on=\"target_id\")\n\n# Display merged dataset\nprint(train_data.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:13:11.370121Z","iopub.execute_input":"2025-03-15T10:13:11.370520Z","iopub.status.idle":"2025-03-15T10:13:11.534210Z","shell.execute_reply.started":"2025-03-15T10:13:11.370492Z","shell.execute_reply":"2025-03-15T10:13:11.533101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Plot sequence lengths\ntrain_sequences['seq_length'] = train_sequences['sequence'].apply(len)\nsns.histplot(train_sequences['seq_length'], bins=30, kde=True)\nplt.xlabel(\"Sequence Length\")\nplt.ylabel(\"Count\")\nplt.title(\"Distribution of RNA Sequence Lengths\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:13:27.803016Z","iopub.execute_input":"2025-03-15T10:13:27.803874Z","iopub.status.idle":"2025-03-15T10:13:28.106629Z","shell.execute_reply.started":"2025-03-15T10:13:27.803724Z","shell.execute_reply":"2025-03-15T10:13:28.105326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Load train sequences\ntrain_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv')\n\n# Compute sequence lengths\ntrain_sequences[\"seq_length\"] = train_sequences[\"sequence\"].apply(len)\n\n# Summary statistics\nprint(train_sequences[\"seq_length\"].describe())\n\n# Boxplot for outliers\nplt.figure(figsize=(10, 5))\nsns.boxplot(x=train_sequences[\"seq_length\"])\nplt.title(\"Boxplot of RNA Sequence Lengths\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:21:39.710067Z","iopub.execute_input":"2025-03-15T10:21:39.710526Z","iopub.status.idle":"2025-03-15T10:21:39.903842Z","shell.execute_reply.started":"2025-03-15T10:21:39.710487Z","shell.execute_reply":"2025-03-15T10:21:39.902606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Load the dataset\ntrain_sequences = pd.read_csv(\"/kaggle/input/stanford-rna-3d-folding/train_sequences.csv\")\n\n# Add a column for sequence length\ntrain_sequences[\"seq_length\"] = train_sequences[\"sequence\"].apply(len)\n\n# Cap sequences at 1000\ntrain_sequences[\"seq_length_capped\"] = train_sequences[\"seq_length\"].apply(lambda x: min(x, 1000))\n\n# Log-transform the sequence lengths (adding 1 to avoid log(0))\ntrain_sequences[\"log_seq_length\"] = np.log1p(train_sequences[\"seq_length_capped\"])\n\n# Plot the transformed distribution\nplt.figure(figsize=(8, 5))\nsns.histplot(train_sequences[\"log_seq_length\"], kde=True, bins=50)\nplt.xlabel(\"Log Sequence Length\")\nplt.ylabel(\"Count\")\nplt.title(\"Log-Transformed Distribution of RNA Sequence Lengths\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:22:36.300663Z","iopub.execute_input":"2025-03-15T10:22:36.301016Z","iopub.status.idle":"2025-03-15T10:22:36.646836Z","shell.execute_reply.started":"2025-03-15T10:22:36.300990Z","shell.execute_reply":"2025-03-15T10:22:36.645716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Assuming df['Sequence_Length'] contains RNA sequence lengths\ndef detect_outliers_iqr(df, column):\n    Q1 = df[column].quantile(0.25)  # 25th percentile\n    Q3 = df[column].quantile(0.75)  # 75th percentile\n    IQR = Q3 - Q1\n    \n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    \n    outliers = df[(df[column] < lower_bound) | (df[column] > upper_bound)]\n    return outliers, lower_bound, upper_bound\n\n# Example Usage\ndf = pd.DataFrame({'Sequence_Length': [3, 22, 39, 86, 100, 4298, 50, 70, 900, 2500]})\noutliers, lb, ub = detect_outliers_iqr(df, 'Sequence_Length')\n\nprint(\"Lower Bound:\", lb)\nprint(\"Upper Bound:\", ub)\nprint(\"Outliers:\\n\", outliers)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:24:35.378016Z","iopub.execute_input":"2025-03-15T10:24:35.378503Z","iopub.status.idle":"2025-03-15T10:24:35.394399Z","shell.execute_reply.started":"2025-03-15T10:24:35.378473Z","shell.execute_reply":"2025-03-15T10:24:35.393012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cleaned = df[(df['Sequence_Length'] >= -945.625) & (df['Sequence_Length'] <=1687.375)]\n\n# Plot the new distribution\nsns.histplot(df_cleaned['Sequence_Length'], bins=50, kde=True)\nplt.xlabel(\"Sequence Length\")\nplt.ylabel(\"Count\")\nplt.title(\"Distribution After Removing Outliers\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:25:35.085994Z","iopub.execute_input":"2025-03-15T10:25:35.086435Z","iopub.status.idle":"2025-03-15T10:25:35.512749Z","shell.execute_reply.started":"2025-03-15T10:25:35.086404Z","shell.execute_reply":"2025-03-15T10:25:35.511474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Apply log transformation (adding 1 to avoid log(0))\ndf[\"Log_Sequence_Length\"] = np.log1p(df[\"Sequence_Length\"])\n\n# Plot the distribution\nplt.figure(figsize=(7,5))\nsns.histplot(df[\"Log_Sequence_Length\"], kde=True, bins=50)\nplt.xlabel(\"Log Sequence Length\")\nplt.ylabel(\"Count\")\nplt.title(\"Log-Transformed Distribution After Removing Outliers\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T10:26:48.851542Z","iopub.execute_input":"2025-03-15T10:26:48.851949Z","iopub.status.idle":"2025-03-15T10:26:49.176743Z","shell.execute_reply.started":"2025-03-15T10:26:48.851919Z","shell.execute_reply":"2025-03-15T10:26:49.175392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}