{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-29T14:39:08.933374Z","iopub.execute_input":"2024-02-29T14:39:08.934073Z","iopub.status.idle":"2024-02-29T14:39:13.857014Z","shell.execute_reply.started":"2024-02-29T14:39:08.934024Z","shell.execute_reply":"2024-02-29T14:39:13.855670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Define the input directory path\ninput_dir = '/kaggle/input'\n\n# List all files under the input directory\nfor dirname, _, filenames in os.walk(input_dir):\n    for filename in filenames:\n        # Construct the full file path\n        file_path = os.path.join(dirname, filename)\n        \n        # Check if the file is a CSV file\n        if file_path.endswith('.csv'):\n            print(f\"Reading data from: {file_path}\")\n            \n            # Read the CSV file into a pandas DataFrame\n            data = pd.read_csv(file_path)\n            \n            # Display the first few rows of the DataFrame\n            print(data.head())\n            print('\\n')\n","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:39:54.431985Z","iopub.execute_input":"2024-02-29T14:39:54.432583Z","iopub.status.idle":"2024-02-29T14:39:55.501189Z","shell.execute_reply.started":"2024-02-29T14:39:54.432548Z","shell.execute_reply":"2024-02-29T14:39:55.500038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import necessary libraries\nimport os\nimport pandas as pd\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:40:21.039971Z","iopub.execute_input":"2024-02-29T14:40:21.040385Z","iopub.status.idle":"2024-02-29T14:40:21.757132Z","shell.execute_reply.started":"2024-02-29T14:40:21.040355Z","shell.execute_reply":"2024-02-29T14:40:21.755925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"**DATASET**\n\n","metadata":{}},{"cell_type":"code","source":"# Define the input directory path\ninput_dir = '/kaggle/input'\n\n# List all files under the input directory\nfor dirname, _, filenames in os.walk(input_dir):\n    for filename in filenames:\n        # Construct the full file path\n        file_path = os.path.join(dirname, filename)\n        \n        # Check if the file is a CSV file\n        if file_path.endswith('.csv'):\n            print(f\"Reading data from: {file_path}\")\n            \n            # Read the CSV file into a pandas DataFrame\n            data = pd.read_csv(file_path)\n            \n            # Display the first few rows of the DataFrame\n            print(data.head())\n            print('\\n')","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:41:19.253970Z","iopub.execute_input":"2024-02-29T14:41:19.254428Z","iopub.status.idle":"2024-02-29T14:41:21.391991Z","shell.execute_reply.started":"2024-02-29T14:41:19.254393Z","shell.execute_reply":"2024-02-29T14:41:21.390821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **DATA ANALYSIS**","metadata":{}},{"cell_type":"code","source":"# Load the dataset\ndata = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\n\n# Display the shape of the dataset (number of rows and columns)\nprint(\"Shape of the dataset:\", data.shape)","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:43:30.991162Z","iopub.execute_input":"2024-02-29T14:43:30.991656Z","iopub.status.idle":"2024-02-29T14:43:31.003233Z","shell.execute_reply.started":"2024-02-29T14:43:30.991621Z","shell.execute_reply":"2024-02-29T14:43:31.001871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the last few rows of the dataset\nprint(\"Last few rows of the dataset:\")\nprint(data.tail())","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:45:05.009635Z","iopub.execute_input":"2024-02-29T14:45:05.010585Z","iopub.status.idle":"2024-02-29T14:45:05.019199Z","shell.execute_reply.started":"2024-02-29T14:45:05.010551Z","shell.execute_reply":"2024-02-29T14:45:05.017989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get information about the dataset\nprint(\"Information about the dataset:\")\nprint(data.info())","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:45:20.202086Z","iopub.execute_input":"2024-02-29T14:45:20.202536Z","iopub.status.idle":"2024-02-29T14:45:20.231576Z","shell.execute_reply.started":"2024-02-29T14:45:20.202503Z","shell.execute_reply":"2024-02-29T14:45:20.230671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Summary statistics for numerical columns\nprint(\"Summary statistics for numerical columns:\")\nprint(data.describe())","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:45:34.561198Z","iopub.execute_input":"2024-02-29T14:45:34.561814Z","iopub.status.idle":"2024-02-29T14:45:34.580976Z","shell.execute_reply.started":"2024-02-29T14:45:34.561771Z","shell.execute_reply":"2024-02-29T14:45:34.579737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for duplicates in the DataFrame\nduplicates = data.duplicated()\n\n# Count the number of duplicate rows\nnum_duplicates = duplicates.sum()\n\nif num_duplicates == 0:\n    print(\"No duplicates found in the DataFrame.\")\nelse:\n    print(\"Number of duplicate rows:\", num_duplicates)\n    print(\"Duplicate rows:\")\n    print(data[duplicates])","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:45:50.842759Z","iopub.execute_input":"2024-02-29T14:45:50.843209Z","iopub.status.idle":"2024-02-29T14:45:50.852751Z","shell.execute_reply.started":"2024-02-29T14:45:50.843175Z","shell.execute_reply":"2024-02-29T14:45:50.851455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the number of missing values in each column\nmissing_values = data.isnull().sum()\n\n# Display the number of missing values for each column\nprint(\"Missing values in each column:\")\nprint(missing_values)","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:46:07.952943Z","iopub.execute_input":"2024-02-29T14:46:07.953369Z","iopub.status.idle":"2024-02-29T14:46:07.967400Z","shell.execute_reply.started":"2024-02-29T14:46:07.953339Z","shell.execute_reply":"2024-02-29T14:46:07.965460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n# Display the count of missing values before imputation\nprint(\"Missing Values Before Imputation:\")\nprint(data.isnull().sum())\n\n# Identify columns with non-numeric data\nnon_numeric_columns = data.select_dtypes(exclude=['number']).columns\n\n# Impute missing values with mean for numeric columns\nnumeric_columns = data.select_dtypes(include=['number']).columns\nimputer_mean = SimpleImputer(strategy='mean')\ndata[numeric_columns] = imputer_mean.fit_transform(data[numeric_columns])\n\n# Impute missing values with most frequent for non-numeric columns\nimputer_mode = SimpleImputer(strategy='most_frequent')\ndata[non_numeric_columns] = imputer_mode.fit_transform(data[non_numeric_columns])\n\n# Display the count of missing values after imputation\nprint(\"\\nMissing Values After Imputation:\")\nprint(data.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-02-29T14:46:27.482303Z","iopub.execute_input":"2024-02-29T14:46:27.483269Z","iopub.status.idle":"2024-02-29T14:46:27.549185Z","shell.execute_reply.started":"2024-02-29T14:46:27.483230Z","shell.execute_reply":"2024-02-29T14:46:27.547940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **DATA REDUCTION**","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}