{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### **Importing Libraries**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-09-11T06:36:28.012404Z","iopub.execute_input":"2023-09-11T06:36:28.012937Z","iopub.status.idle":"2023-09-11T06:36:28.350718Z","shell.execute_reply.started":"2023-09-11T06:36:28.012893Z","shell.execute_reply":"2023-09-11T06:36:28.349288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-11T06:36:28.353483Z","iopub.execute_input":"2023-09-11T06:36:28.353986Z","iopub.status.idle":"2023-09-11T06:37:38.818457Z","shell.execute_reply.started":"2023-09-11T06:36:28.353955Z","shell.execute_reply":"2023-09-11T06:37:38.816928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-11T06:37:38.820477Z","iopub.execute_input":"2023-09-11T06:37:38.821154Z","iopub.status.idle":"2023-09-11T06:37:38.859457Z","shell.execute_reply.started":"2023-09-11T06:37:38.821093Z","shell.execute_reply":"2023-09-11T06:37:38.858248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dataset is too large","metadata":{}},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-11T06:37:38.861380Z","iopub.execute_input":"2023-09-11T06:37:38.861664Z","iopub.status.idle":"2023-09-11T06:37:38.868112Z","shell.execute_reply.started":"2023-09-11T06:37:38.861642Z","shell.execute_reply":"2023-09-11T06:37:38.867178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# iterating the columns\nfor col in df.columns:\n    print(col)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T06:40:31.880464Z","iopub.execute_input":"2023-09-11T06:40:31.881607Z","iopub.status.idle":"2023-09-11T06:40:31.891030Z","shell.execute_reply.started":"2023-09-11T06:40:31.881534Z","shell.execute_reply":"2023-09-11T06:40:31.889916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So I am taking a **sample** from the whole dataset just for initial exploration befor working on the entire dataset.\nReducing fraction to 0.2.\nNow I am using 20% of the dataset","metadata":{}},{"cell_type":"code","source":"sample_df = df.sample(frac=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T07:00:33.974327Z","iopub.execute_input":"2023-09-11T07:00:33.974716Z","iopub.status.idle":"2023-09-11T07:00:35.523891Z","shell.execute_reply.started":"2023-09-11T07:00:33.974677Z","shell.execute_reply":"2023-09-11T07:00:35.522726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-11T07:00:35.526342Z","iopub.execute_input":"2023-09-11T07:00:35.527453Z","iopub.status.idle":"2023-09-11T07:00:35.533327Z","shell.execute_reply.started":"2023-09-11T07:00:35.527416Z","shell.execute_reply":"2023-09-11T07:00:35.532440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-11T07:00:36.425016Z","iopub.execute_input":"2023-09-11T07:00:36.425699Z","iopub.status.idle":"2023-09-11T07:00:36.450774Z","shell.execute_reply.started":"2023-09-11T07:00:36.425660Z","shell.execute_reply":"2023-09-11T07:00:36.450001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_null_values = sample_df.isnull().sum()\ncheck_null_values","metadata":{"execution":{"iopub.status.busy":"2023-09-11T07:00:38.028883Z","iopub.execute_input":"2023-09-11T07:00:38.029590Z","iopub.status.idle":"2023-09-11T07:00:38.283384Z","shell.execute_reply.started":"2023-09-11T07:00:38.029534Z","shell.execute_reply":"2023-09-11T07:00:38.282065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now Finding the number of columns with null values","metadata":{}},{"cell_type":"code","source":"# Filter columns with null values (count > 0)\ncolumns_with_null = check_null_values[check_null_values > 0]\n\n# Get the number of columns with null values\nnum_columns_with_null = len(columns_with_null)\n\nprint(f\"Number of columns with null values: {num_columns_with_null}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-11T07:00:57.002095Z","iopub.execute_input":"2023-09-11T07:00:57.003149Z","iopub.status.idle":"2023-09-11T07:00:57.010387Z","shell.execute_reply.started":"2023-09-11T07:00:57.003084Z","shell.execute_reply":"2023-09-11T07:00:57.009171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 412 columns have null values","metadata":{}},{"cell_type":"markdown","source":"removing those columns from the Dataframe","metadata":{}},{"cell_type":"code","source":"sample_df = sample_df.dropna(axis=1, how='all')","metadata":{"execution":{"iopub.status.busy":"2023-09-11T07:22:06.295800Z","iopub.execute_input":"2023-09-11T07:22:06.296427Z","iopub.status.idle":"2023-09-11T07:22:08.732333Z","shell.execute_reply.started":"2023-09-11T07:22:06.296392Z","shell.execute_reply":"2023-09-11T07:22:08.730719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df","metadata":{"execution":{"iopub.status.busy":"2023-09-11T07:22:11.512684Z","iopub.execute_input":"2023-09-11T07:22:11.513060Z","iopub.status.idle":"2023-09-11T07:22:11.719165Z","shell.execute_reply.started":"2023-09-11T07:22:11.513033Z","shell.execute_reply":"2023-09-11T07:22:11.717716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now we are left with 285 columns","metadata":{}},{"cell_type":"code","source":"sample_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-11T07:22:21.385479Z","iopub.execute_input":"2023-09-11T07:22:21.385844Z","iopub.status.idle":"2023-09-11T07:22:21.393050Z","shell.execute_reply.started":"2023-09-11T07:22:21.385817Z","shell.execute_reply":"2023-09-11T07:22:21.391711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-09-11T08:03:03.118987Z","iopub.execute_input":"2023-09-11T08:03:03.120197Z","iopub.status.idle":"2023-09-11T08:03:03.669257Z","shell.execute_reply.started":"2023-09-11T08:03:03.120150Z","shell.execute_reply":"2023-09-11T08:03:03.668538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dataset contains 297 float values, 2, int values and 4 objects**","metadata":{}},{"cell_type":"code","source":"sample_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-09-11T08:06:28.717610Z","iopub.execute_input":"2023-09-11T08:06:28.717971Z","iopub.status.idle":"2023-09-11T08:06:28.746725Z","shell.execute_reply.started":"2023-09-11T08:06:28.717943Z","shell.execute_reply":"2023-09-11T08:06:28.745760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-09-11T08:05:45.507365Z","iopub.execute_input":"2023-09-11T08:05:45.507744Z","iopub.status.idle":"2023-09-11T08:05:48.858750Z","shell.execute_reply.started":"2023-09-11T08:05:45.507716Z","shell.execute_reply":"2023-09-11T08:05:48.857441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Value Counts of experiment_type**","metadata":{}},{"cell_type":"code","source":"sample_df['experiment_type'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-09-11T08:04:34.954540Z","iopub.execute_input":"2023-09-11T08:04:34.955254Z","iopub.status.idle":"2023-09-11T08:04:34.981784Z","shell.execute_reply.started":"2023-09-11T08:04:34.955204Z","shell.execute_reply":"2023-09-11T08:04:34.980462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Total Number of Sequence ids in sample df**","metadata":{}},{"cell_type":"code","source":"sample_df['sequence_id'].count()","metadata":{"execution":{"iopub.status.busy":"2023-09-11T08:01:11.327805Z","iopub.execute_input":"2023-09-11T08:01:11.328250Z","iopub.status.idle":"2023-09-11T08:01:11.398556Z","shell.execute_reply.started":"2023-09-11T08:01:11.328185Z","shell.execute_reply":"2023-09-11T08:01:11.397489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Number of different sequence_ids:**","metadata":{}},{"cell_type":"code","source":"sequence_id_groups = sample_df['sequence_id'].nunique()\n\nprint(\"Number of unique groups (sequence_id):\", sequence_id_groups)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T08:00:37.484653Z","iopub.execute_input":"2023-09-11T08:00:37.485111Z","iopub.status.idle":"2023-09-11T08:00:37.720483Z","shell.execute_reply.started":"2023-09-11T08:00:37.485078Z","shell.execute_reply":"2023-09-11T08:00:37.718799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}