{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[],"machine_shape":"hm","gpuType":"A100"},"accelerator":"GPU","kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Install PyTorch Geometric and dependencies\n!pip install torch-scatter -f https://data.pyg.org/whl/torch-2.0.0+cpu.html\n!pip install torch-sparse -f https://data.pyg.org/whl/torch-2.0.0+cpu.html\n!pip install torch-geometric\n\n","metadata":{"id":"cYJcKY6QunI0","outputId":"6143e21f-e23b-4a28-913c-83792aeb9b3e","trusted":true,"execution":{"iopub.status.busy":"2025-04-25T11:42:13.799381Z","iopub.execute_input":"2025-04-25T11:42:13.799670Z","iopub.status.idle":"2025-04-25T11:42:22.892921Z","shell.execute_reply.started":"2025-04-25T11:42:13.799643Z","shell.execute_reply":"2025-04-25T11:42:22.891882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install torch-scatter -f https://data.pyg.org/whl/torch-2.1.0+cpu.html\n!pip install torch-sparse -f https://data.pyg.org/whl/torch-2.1.0+cpu.html\n","metadata":{"id":"f0gcWiNJbEqp","outputId":"83851e09-a409-4633-9f53-95e267386e3f","trusted":true,"execution":{"iopub.status.busy":"2025-04-25T11:42:50.358262Z","iopub.execute_input":"2025-04-25T11:42:50.358578Z","iopub.status.idle":"2025-04-25T11:42:56.380952Z","shell.execute_reply.started":"2025-04-25T11:42:50.358555Z","shell.execute_reply":"2025-04-25T11:42:56.380015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install additional visualization packages\n!pip install matplotlib missingno seaborn plotly networkx\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport missingno as msno\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nimport networkx as nx\nfrom networkx.drawing.nx_agraph import graphviz_layout","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T11:43:00.715636Z","iopub.execute_input":"2025-04-25T11:43:00.716394Z","iopub.status.idle":"2025-04-25T11:43:03.812092Z","shell.execute_reply.started":"2025-04-25T11:43:00.716365Z","shell.execute_reply":"2025-04-25T11:43:03.811368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and visualize initial data\ndf = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')\n\nprint(\"\\n=== Initial Data Preview ===\")\nprint(df.head())\n\nprint(\"\\n=== Data Statistics ===\")\nprint(df.describe())\n\nprint(\"\\n=== Missing Values ===\")\nprint(df.isnull().sum())\n\n# Missing data visualization\nplt.figure(figsize=(10, 6))\nmsno.matrix(df)\nplt.title('Missing Value Matrix', fontsize=16)\nplt.show()\n\n# Distribution of coordinates (non-missing values)\ncoord_cols = ['x_1', 'y_1', 'z_1']\ndf[coord_cols].dropna().hist(bins=50, figsize=(15, 5), layout=(1, 3))\nplt.suptitle('Coordinate Distributions', y=1.05)\nplt.tight_layout()\nplt.show()\n\n# 3D scatter plot of coordinates (sample)\nsample_df = df.dropna().sample(n=1000, random_state=42)\nfig = px.scatter_3d(sample_df, x='x_1', y='y_1', z='z_1', \n                    color='resname', title='3D RNA Structure (Sample)')\nfig.update_layout(scene=dict(aspectmode=\"cube\"))\nfig.show()","metadata":{"id":"US1tGoqzg70K","outputId":"80eb4b41-ae3b-4234-faeb-b3f6431d02fe","trusted":true,"execution":{"iopub.status.busy":"2025-04-25T11:43:35.899794Z","iopub.execute_input":"2025-04-25T11:43:35.900157Z","iopub.status.idle":"2025-04-25T11:43:37.591989Z","shell.execute_reply.started":"2025-04-25T11:43:35.900132Z","shell.execute_reply":"2025-04-25T11:43:37.591234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom pathlib import Path\n\n# Load the dataset\ninput_path = Path('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')\ndf = pd.read_csv(input_path)\n\n# Define maximum gap length for interpolation\nMAX_GAP = 5\n\ndef interpolate_group(group):\n    grp = group.copy()\n    # Identify where any coordinate is missing\n    missing_mask = grp[['x_1', 'y_1', 'z_1']].isna().any(axis=1)\n\n    # Identify continuous missing segments\n    gaps = []\n    current_gap = []\n    for idx, miss in zip(grp.index, missing_mask):\n        if miss:\n            current_gap.append(idx)\n        else:\n            if current_gap:\n                gaps.append(current_gap)\n                current_gap = []\n    if current_gap:\n        gaps.append(current_gap)\n\n    # Fill large gaps with group mean\n    mean_vals = grp[['x_1', 'y_1', 'z_1']].mean()\n    for gap in gaps:\n        if len(gap) > MAX_GAP:\n            grp.loc[gap, ['x_1', 'y_1', 'z_1']] = mean_vals.values\n\n    # Interpolate remaining missing values by 'resid'\n    grp = grp.set_index('resid')\n    grp[['x_1', 'y_1', 'z_1']] = grp[['x_1', 'y_1', 'z_1']].interpolate(\n        method='index', limit=MAX_GAP, limit_direction='both'\n    )\n    return grp.reset_index()\n\n# Apply interpolation per ID (modified to avoid warning)\ndf_imputed = (\n    df.groupby('ID', group_keys=False)\n      .apply(lambda x: interpolate_group(x))\n      .reset_index(drop=True)\n)\n\n# Final fallback: fill any remaining NaNs with global mean\ndf_imputed[['x_1', 'y_1', 'z_1']] = df_imputed[['x_1', 'y_1', 'z_1']].fillna(\n    df[['x_1', 'y_1', 'z_1']].mean()\n)\n\n# Check remaining missing\nremaining_na = df_imputed[['x_1', 'y_1', 'z_1']].isna().sum()\nprint(\"Remaining missing after final cleaning:\", remaining_na.to_dict())\n\n# Save the cleaned dataset to a writable directory\noutput_path = Path('/kaggle/working/train_labels_cleaned.csv')\ndf_imputed.to_csv(output_path, index=False)\nprint(\"Cleaned data saved to:\", output_path)\n","metadata":{"id":"9lzbe7nNmesR","outputId":"5208f833-812f-44ca-adf0-07e8c738d1de","trusted":true,"execution":{"iopub.status.busy":"2025-04-25T11:02:20.439350Z","iopub.execute_input":"2025-04-25T11:02:20.439646Z","iopub.status.idle":"2025-04-25T11:09:00.390181Z","shell.execute_reply.started":"2025-04-25T11:02:20.439625Z","shell.execute_reply":"2025-04-25T11:09:00.389436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and visualize initial data\ndf = pd.read_csv('/kaggle/working/train_labels_cleaned.csv')\n\nprint(\"\\n=== Initial Data Preview ===\")\nprint(df.head())\n\nprint(\"\\n=== Data Statistics ===\")\nprint(df.describe())\n\nprint(\"\\n=== Missing Values ===\")\nprint(df.isnull().sum())\n\n# Missing data visualization\nplt.figure(figsize=(10, 6))\nmsno.matrix(df)\nplt.title('Missing Value Matrix', fontsize=16)\nplt.show()\n\n# Distribution of coordinates (non-missing values)\ncoord_cols = ['x_1', 'y_1', 'z_1']\ndf[coord_cols].dropna().hist(bins=50, figsize=(15, 5), layout=(1, 3))\nplt.suptitle('Coordinate Distributions', y=1.05)\nplt.tight_layout()\nplt.show()\n\n# 3D scatter plot of coordinates (sample)\nsample_df = df.dropna().sample(n=1000, random_state=42)\nfig = px.scatter_3d(sample_df, x='x_1', y='y_1', z='z_1', \n                    color='resname', title='3D RNA Structure (Sample)')\nfig.update_layout(scene=dict(aspectmode=\"cube\"))\nfig.show()","metadata":{"id":"UddM_XUCqQGF","outputId":"ca1fd9b5-3a20-495f-d4df-f98494a0bee3","trusted":true,"execution":{"iopub.status.busy":"2025-04-25T11:43:50.572359Z","iopub.execute_input":"2025-04-25T11:43:50.572632Z","iopub.status.idle":"2025-04-25T11:43:51.947445Z","shell.execute_reply.started":"2025-04-25T11:43:50.572614Z","shell.execute_reply":"2025-04-25T11:43:51.946716Z"}},"outputs":[],"execution_count":null}]}