{"cells":[{"cell_type":"code","execution_count":1,"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"outputs":[],"source":"import numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom IPython.display import display, Markdown\nimport scipy.stats as stats\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Decide between local or kaggle cloud storage         \nKAGGLE_ENV = 'kaggle' in os.listdir('/')\ndata_path = '/kaggle/input' if KAGGLE_ENV else '../kaggle/input'\n    \n    \nfor dirname, _, filenames in os.walk(data_path):\n    for filename in filenames:\n        print(os.path.join(dirname, filename)) "},{"cell_type":"markdown","metadata":{},"source":"# Introduction\nUsually I start with a EDA script. This time it is kinda the same but for this competition there is more. We need to understand the world of RNA and fully understand first the data before I play with the data..\nIn this exploratory data analysis (EDA), we aim to gain a deeper understanding of the dataset by performing the following key steps:"},{"cell_type":"markdown","metadata":{},"source":"# Load Data"},{"cell_type":"code","execution_count":2,"metadata":{},"outputs":[],"source":"file_paths = {\n    \"df_sample_submission_sunny\": \"../kaggle/input/standford-rna-3d-folding-sunny/sample_submission.csv\",\n    \"df_test_sequences_sunny\": \"../kaggle/input/standford-rna-3d-folding-sunny/test_sequences.csv\",\n    \"df_train_labels_sunny\": \"../kaggle/input/standford-rna-3d-folding-sunny/train_labels.csv\",\n    \"df_train_sequences_sunny\": \"../kaggle/input/standford-rna-3d-folding-sunny/train_sequences.csv\",\n    \"df_validation_labels\": \"../kaggle/input/stanford-rna-3d-folding/validation_labels.csv\",\n    \"df_sample_submission\": \"../kaggle/input/stanford-rna-3d-folding/sample_submission.csv\",\n    \"df_test_sequences\": \"../kaggle/input/stanford-rna-3d-folding/test_sequences.csv\",\n    \"df_validation_sequences\": \"../kaggle/input/stanford-rna-3d-folding/validation_sequences.csv\",\n    \"df_train_labels\": \"../kaggle/input/stanford-rna-3d-folding/train_labels.csv\",\n    \"df_train_sequences\": \"../kaggle/input/stanford-rna-3d-folding/train_sequences.csv\"\n}\n\nfor var_name, path in file_paths.items():\n    try:\n        globals()[var_name] = pd.read_csv(path)\n        print(f\"{var_name} load, {globals()[var_name].shape[0]} rows, {globals()[var_name].shape[1]} columns.\")\n    except FileNotFoundError:\n        print(f\"file not found: {path}\")\n    except Exception as e:\n        print(f\"error .. {path}: {e}\")\n"},{"cell_type":"markdown","metadata":{},"source":"# Quick Overview"},{"cell_type":"code","execution_count":17,"metadata":{},"outputs":[],"source":"df_train_sequences.head(5)"},{"cell_type":"code","execution_count":19,"metadata":{},"outputs":[],"source":"# print('df_sample_submission_sunny')\n# display(df_sample_submission_sunny.head())\n# print('df_test_sequences_sunny')\n# display(df_test_sequences_sunny.head())\ndisplay('df_train_labels_sunny')\ndisplay(df_train_labels_sunny.head(100))\ndisplay('df_train_sequences_sunny')\ndisplay(df_train_sequences_sunny.head(100))\n# display('df_validation_labels')\n# display(df_validation_labels.head())\n# display('df_sample_submission')\n# display(df_sample_submission.head())\n# display('df_test_sequences')\n# display(df_test_sequences.head())\n# display('df_validation_sequences')\n# display(df_validation_sequences.head())\n# display('df_train_labels')\n# display(df_train_labels.head())\n# display('df_train_sequences')\n# display(df_train_sequences.head())\n"},{"cell_type":"code","execution_count":89,"metadata":{},"outputs":[],"source":"original_data.head()"},{"cell_type":"code","execution_count":90,"metadata":{},"outputs":[],"source":"sample_submission.head(100)"},{"cell_type":"code","execution_count":91,"metadata":{},"outputs":[],"source":"original_data.head()"},{"cell_type":"markdown","metadata":{},"source":"# Overview"},{"cell_type":"code","execution_count":92,"metadata":{},"outputs":[],"source":"def data_overview(data, target):\n    # Overview\n    display(Markdown(\"## Data Overview\"))\n    \n    display(Markdown(\"### General Information\"))\n    display(Markdown(f\"- Number of rows and columns: {data.shape[0]} x {data.shape[1]}\"))\n    display(Markdown(\"- Column names:\"))\n    display(list(data.columns))\n\n    display(Markdown(\"### Data Types & Missing Values\"))\n    missing = data.isnull().sum()\n    dtypes = pd.DataFrame(data.dtypes, columns=[\"Data Type\"])\n    missing_df = pd.DataFrame(missing, columns=[\"Missing Values\"])\n    overview_df = dtypes.join(missing_df)\n    display(overview_df.style.background_gradient(cmap=\"coolwarm\"))\n\n    display(Markdown(\"### Classic head of Data\"))\n    display(data.head().style.set_properties(**{\"background-color\": \"#f5f5f5\"}))\n\n    display(Markdown(\"### Statistical Summary (describe)\"))\n    display(data.describe().T.style.background_gradient(cmap=\"viridis\"))\n\n    # Target variable analysis\n    if target is not None:\n        display(Markdown(f\"## Target Variable: `{target}`\"))\n        sns.set_style(\"whitegrid\")  \n        sns.set_palette(\"viridis\")   \n\n        fig, ax = plt.subplots(1, 2, figsize=(14, 5))\n\n        #   Absolute frequency barplot\n        sns.barplot(x=data[target].value_counts().index, \n                y=data[target].value_counts(), \n                ax=ax[0])  \n\n        ax[0].set_title(\"Absolute Frequency\", fontsize=12, fontweight=\"bold\")\n        ax[0].set_ylabel(\"Count\")\n        ax[0].set_xlabel(target)\n        ax[0].grid(axis=\"y\", linestyle=\"--\", alpha=0.5)  \n\n        # Percentage distribution barplot\n        sns.barplot(x=data[target].value_counts().index, \n                    y=data[target].value_counts(normalize=True), \n                    ax=ax[1])  \n\n        ax[1].set_title(\"Percentage Distribution\", fontsize=12, fontweight=\"bold\")\n        ax[1].set_ylabel(\"Percentage\")\n        ax[1].set_xlabel(target)\n        ax[1].grid(axis=\"y\", linestyle=\"--\", alpha=0.5)\n\n    \n\n        for spine in [\"top\", \"right\"]:\n            ax[0].spines[spine].set_visible(False)\n            ax[1].spines[spine].set_visible(False)\n\n    plt.tight_layout()\n    plt.show()"},{"cell_type":"markdown","metadata":{},"source":"# Quick PreProcessing about some features"},{"cell_type":"markdown","metadata":{},"source":"# Optional Point: Concat the data"},{"cell_type":"markdown","metadata":{},"source":"# Feature Analyse"},{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":"def visualize_feature_attributes(df, target=None):\n    \"\"\" Visualizes numeric and categorical features \"\"\"\n\n    # Get Numeric & Categorical Features\n    numeric_features, categorical_features =get_categorical_numerical_features(df)\n\n    # Numeric Features\n    if numeric_features:\n        display(Markdown(\"## Numeric Feature Attributes\"))\n        for col in numeric_features:\n            if col != target:\n                plot_numeric_feature(df, col, target)\n    else:\n        print(\"No numeric features found.\")\n\n    # Categorical Features\n    if categorical_features:\n        display(Markdown(\"## Categorical Feature Attributes\"))\n        for col in categorical_features:\n            if col != target:\n                if df[col].nunique() > 10:\n                    df = reduce_categories(df, col, top_n=15)\n                plot_categorical_feature(df, col, target)\n    else:\n        print(\"No categorical features found.\")\n\n\ndef plot_numeric_feature(df, col, target):\n    \"\"\" Plots Histogram, Boxplot, and Violinplot for a numeric feature \"\"\"\n    fig, axes = plt.subplots(1, 3, figsize=(20, 5))\n\n    sns.histplot(df[col], ax=axes[0], kde=True)\n    axes[0].set_title(f\"Distribution of {col}\", fontweight=\"bold\")\n\n    sns.boxplot(x=df[col], ax=axes[1])\n    axes[1].set_title(f\"Boxplot of {col}\", fontweight=\"bold\")\n\n    if target and target in df.columns and df[target].nunique() == 2:\n        sns.violinplot(x=df[target], y=df[col], ax=axes[2], split=True)\n    elif target and target in df.columns:\n        sns.violinplot(x=df[target], y=df[col], ax=axes[2], split=False)\n    else:\n        sns.violinplot(y=df[col], ax=axes[2])\n\n    axes[2].set_title(f\"Violinplot of {col} by {target}\", fontweight=\"bold\")\n\n    plt.tight_layout()\n    plt.show()\n\n\ndef plot_categorical_feature(df, col, target):\n    \"\"\" Plots Countplot, Hue-Countplot, and Barplot (if target is numeric) for a categorical feature \"\"\"\n    fig, axes = plt.subplots(1, 3, figsize=(20, 5))\n\n    sns.countplot(x=df[col], ax=axes[0])\n    axes[0].set_title(f\"Countplot of {col}\", fontweight=\"bold\")\n    axes[0].tick_params(axis='x', rotation=45)\n\n    if target in df.columns:\n        sns.countplot(x=df[col], hue=df[target], ax=axes[1])\n        axes[1].set_title(f\"Countplot of {col} by {target}\", fontweight=\"bold\")\n        axes[1].tick_params(axis='x', rotation=45)\n\n    if target in df.columns and df[target].dtype in [np.float64, np.int64]:\n        sns.barplot(x=df[col], y=df[target], ax=axes[2], estimator=np.mean, errorbar='sd')\n        axes[2].set_title(f\"Mean {target} by {col}\", fontweight=\"bold\")\n    else:\n        axes[2].remove()  \n\n    plt.tight_layout()\n    plt.show()\n    \n\ndef reduce_categories(df, col, top_n):\n    \"\"\" Shows only the categories with highes numbers, seldoms are shown with \"others\" \"\"\"\n    top_categories = df[col].value_counts().nlargest(top_n).index\n    df[col] = df[col].apply(lambda x: x if x in top_categories else 'Other')\n    return df\n\ndef get_categorical_numerical_features(df):\n    # Get Numeric & Categorical Features\n    numeric_features = df.select_dtypes(include=[np.number]).columns.tolist()\n    categorical_features = df.select_dtypes(include=['object', 'category']).columns.tolist()\n    return numeric_features, categorical_features\n\n#visualize_feature_attributes(train, target=target_feature)"},{"cell_type":"markdown","metadata":{},"source":"# Correlation Matrix (Numerical values)"},{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":"# Get Numeric & Categorical Features\n# numeric_features, categorical_features = get_categorical_numerical_features(train)\n# sns.heatmap(train[numeric_features].corr(), annot=True, cmap='coolwarm')"},{"cell_type":"markdown","metadata":{},"source":"# Analyse categorical features"},{"cell_type":"markdown","metadata":{},"source":"# Correlation Matrix Categorical Features"},{"cell_type":"markdown","metadata":{},"source":"# Merge categorical features\nI will wait before i will drop features"},{"cell_type":"markdown","metadata":{},"source":"# Analyze\n## Numerical Features\n## Categorical Features\n"},{"cell_type":"markdown","metadata":{},"source":"# Save CSV Files as Kaggle Datasets"},{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":"# if KAGGLE_ENV:\n#     path = '/kaggle/input/' + competition_name + '-train-concat/'\n#     if not os.path.exists(path):\n#         os.makedirs(path)\n#     train.to_csv('/kaggle/input/' + competition_name + '-train-concat/' + competition_name + '-train-concat.csv', index=False)\n# else:\n#     path = '../kaggle/input/' + competition_name + '-train-concat/'\n#     if not os.path.exists(path):\n#         os.makedirs(path)\n#     train.to_csv('../kaggle/input/' + competition_name + '-train-concat/' + competition_name + '-train-concat.csv', index=False)"},{"cell_type":"code","execution_count":null,"metadata":{},"outputs":[],"source":"# if KAGGLE_ENV:\n#     path = '/kaggle/input/' + competition_name + '-test-concat/'\n#     if not os.path.exists(path):\n#         os.makedirs(path)\n#     test.to_csv('/kaggle/input/' + competition_name + '-test-concat/' + competition_name + '-test-concat.csv', index=False)\n# else:\n#     path = '../kaggle/input/' + competition_name + '-test-concat/'\n#     if not os.path.exists(path):\n#         os.makedirs(path)\n#     test.to_csv('../kaggle/input/' + competition_name + '-test-concat/' + competition_name + '-test-concat.csv', index=False)"},{"cell_type":"code","execution_count":107,"metadata":{},"outputs":[],"source":"train.head()"},{"cell_type":"code","execution_count":108,"metadata":{},"outputs":[],"source":"test.head()"}],"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"databundleVersionId":10008389,"sourceId":84895,"sourceType":"competition"}],"isGpuEnabled":false,"isInternetEnabled":true,"language":"python","sourceType":"notebook"},"kernelspec":{"display_name":"base","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.5"}},"nbformat":4,"nbformat_minor":4}