{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\n# Step 1: Define the file paths\n# The file paths will be available under the directory /kaggle/input/competition-name/\ntrain_file_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'  # Update with the actual file name\ntest_file_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'    # Update with the actual file name\n\n# Step 2: Load the data into pandas DataFrames\ntrain_data = pd.read_csv(train_file_path)\ntest_data = pd.read_csv(test_file_path)\n\n# Step 3: Display the first few rows of the training and testing datasets\nprint(\"Training Data:\")\nprint(train_data.head())\n\nprint(\"\\nTest Data:\")\nprint(test_data.head())\n\n# Step 4: Display sample input features and output labels\n# Assuming the target column is named 'target_column_name', update accordingly\ninput_features = train_data.drop(columns=['target_column_name'])  # Replace 'target_column_name' with the actual target column\noutput_labels = train_data['target_column_name']                  # Replace 'target_column_name' with the actual target column\n\nprint(\"\\nInput Features Sample:\")\nprint(input_features.head())\n\nprint(\"\\nOutput Labels Sample:\")\nprint(output_labels.head())\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries\nimport pandas as pd\n\n# Define the path to the dataset\ninput_path = '/kaggle/input/child-mind-institute-problematic-internet-use'\n\n# Load the dataset\ntrain_data = pd.read_csv(input_path + 'train.csv')\ntest_data = pd.read_csv(input_path + 'test.csv')\n\n# Display the first few rows of the training data\nprint(\"Training Data:\")\nprint(train_data.head())\n\n# Display the first few rows of the test data\nprint(\"\\nTest Data:\")\nprint(test_data.head())\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries\nimport pandas as pd\n\n# Define the paths to the datasets\n#input_path = '/kaggle/input/child-mind-institute-problematic-intern/'\ntrain_series_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\ntest_series_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\n\n# Load the datasets\ntrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ntrain_series = pd.read_parquet(train_series_path)\ntest_series = pd.read_parquet(test_series_path)\n\n# Display the first few rows of the training data\nprint(\"Training Data:\")\nprint(train_data.head())\n\n# Display the first few rows of the test data\nprint(\"\\nTest Data:\")\nprint(test_data.head())\n\n# Display the first few rows of the training series data\nprint(\"\\nTraining Series Data:\")\nprint(train_series.head())\n\n# Display the first few rows of the test series data\nprint(\"\\nTest Series Data:\")\nprint(test_series.head())\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Step 1: Define file paths\nseries_train_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\nseries_test_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\ntrain_file_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntest_file_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\ndata_dictionary_path = '/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv'\nsample_submission_path = '/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv'\n\n# Step 2: Load the Parquet files into pandas DataFrames\nseries_train = pd.read_parquet(series_train_path)\nseries_test = pd.read_parquet(series_test_path)\n\n# Step 3: Load the CSV files into pandas DataFrames\ntrain_data = pd.read_csv(train_file_path)\ntest_data = pd.read_csv(test_file_path)\ndata_dictionary = pd.read_csv(data_dictionary_path)\nsample_submission = pd.read_csv(sample_submission_path)\n\n# Step 4: Display a few rows from each DataFrame to understand the structure\nprint(\"Series Train Data:\")\nprint(series_train.head())\n\nprint(\"\\nSeries Test Data:\")\nprint(series_test.head())\n\nprint(\"\\nTraining Data:\")\nprint(train_data.head())\n\nprint(\"\\nTest Data:\")\nprint(test_data.head())\n\nprint(\"\\nData Dictionary:\")\nprint(data_dictionary.head())\n\nprint(\"\\nSample Submission:\")\nprint(sample_submission.head())\n\n# Step 5: Show example input features and target labels from the training data\n# Assuming 'target_column_name' is the actual name of the target column in train.csv\ninput_features = train_data.drop(columns=['target_column_name'])  # Replace 'target_column_name' with actual target column\noutput_labels = train_data['target_column_name']                  # Replace 'target_column_name' with actual target column\n\nprint(\"\\nInput Features Sample:\")\nprint(input_features.head())\n\nprint(\"\\nOutput Labels Sample:\")\nprint(output_labels.head())\n\n# Step 6: Explore 'series_train' DataFrame - Checking unique patient ids and sample data for one patient\nunique_ids = series_train['id'].unique()\nprint(f\"\\nNumber of unique patient IDs in training series: {len(unique_ids)}\")\nprint(f\"Sample data for patient ID {unique_ids[0]}:\\n\")\nprint(series_train[series_train['id'] == unique_ids[0]].head())\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Step 1: Define file paths\nseries_train_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\nseries_test_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\ntrain_file_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntest_file_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\ndata_dictionary_path = '/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv'\nsample_submission_path = '/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv'\n\n\n# Step 2: Check if files exist before loading\ndef file_exists(file_path):\n    return os.path.exists(file_path)\n\ndef load_data(file_path, file_type=\"csv\"):\n    try:\n        if file_type == \"csv\":\n            return pd.read_csv(file_path)\n        elif file_type == \"parquet\":\n            return pd.read_parquet(file_path)\n    except Exception as e:\n        print(f\"Error loading {file_path}: {e}\")\n        return None\n\n# Load data with checks\nif file_exists(series_train_path):\n    series_train = load_data(series_train_path, file_type=\"parquet\")\nelse:\n    print(\"Series train file does not exist.\")\n\nif file_exists(series_test_path):\n    series_test = load_data(series_test_path, file_type=\"parquet\")\nelse:\n    print(\"Series test file does not exist.\")\n\nif file_exists(train_file_path):\n    train_data = load_data(train_file_path, file_type=\"csv\")\nelse:\n    print(\"Training CSV file does not exist.\")\n\nif file_exists(test_file_path):\n    test_data = load_data(test_file_path, file_type=\"csv\")\nelse:\n    print(\"Test CSV file does not exist.\")\n\nif file_exists(data_dictionary_path):\n    data_dictionary = load_data(data_dictionary_path, file_type=\"csv\")\nelse:\n    print(\"Data dictionary file does not exist.\")\n\nif file_exists(sample_submission_path):\n    sample_submission = load_data(sample_submission_path, file_type=\"csv\")\nelse:\n    print(\"Sample submission file does not exist.\")\n\n# Step 3: Display a few rows from each DataFrame if they loaded successfully\ndef display_head(df, name):\n    if df is not None:\n        print(f\"\\n{name}:\")\n        print(df.head())\n    else:\n        print(f\"{name} could not be loaded.\")\n\ndisplay_head(series_train, \"Series Train Data\")\ndisplay_head(series_test, \"Series Test Data\")\ndisplay_head(train_data, \"Training Data\")\ndisplay_head(test_data, \"Test Data\")\ndisplay_head(data_dictionary, \"Data Dictionary\")\ndisplay_head(sample_submission, \"Sample Submission\")\n\n# Step 4: Handle input features and output labels (if train_data is loaded successfully)\nif train_data is not None:\n    try:\n        # Replace 'target_column_name' with the actual name of the target column\n        target_column = 'target_column_name'\n        if target_column in train_data.columns:\n            input_features = train_data.drop(columns=[target_column])\n            output_labels = train_data[target_column]\n\n            print(\"\\nInput Features Sample:\")\n            print(input_features.head())\n\n            print(\"\\nOutput Labels Sample:\")\n            print(output_labels.head())\n        else:\n            print(f\"Target column '{target_column}' not found in training data.\")\n    except Exception as e:\n        print(f\"Error processing input and output features: {e}\")\nelse:\n    print(\"Training data is not available to process input and output features.\")\n\n# Step 5: Explore 'series_train' DataFrame if it is loaded\nif series_train is not None:\n    try:\n        unique_ids = series_train['id'].unique()\n        print(f\"\\nNumber of unique patient IDs in training series: {len(unique_ids)}\")\n        print(f\"Sample data for patient ID {unique_ids[0]}:\\n\")\n        print(series_train[series_train['id'] == unique_ids[0]].head())\n    except Exception as e:\n        print(f\"Error exploring series_train: {e}\")\nelse:\n    print(\"Series train data is not available for exploration.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-23T19:51:36.941816Z","iopub.execute_input":"2024-10-23T19:51:36.942536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Step 1: Define file paths\nseries_train_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\nseries_test_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\ntrain_file_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntest_file_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\ndata_dictionary_path = '/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv'\nsample_submission_path = '/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv'\n\n\n# Step 2: Helper function to reduce memory usage by downcasting data types\ndef reduce_memory_usage(df):\n    for col in df.select_dtypes(include=['float']):\n        df[col] = pd.to_numeric(df[col], downcast='float')\n    for col in df.select_dtypes(include=['int']):\n        df[col] = pd.to_numeric(df[col], downcast='integer')\n    return df\n\n# Step 3: Read data in chunks for large CSV files\ndef load_csv_in_chunks(file_path, chunksize=10000, usecols=None):\n    chunk_list = []\n    try:\n        for chunk in pd.read_csv(file_path, chunksize=chunksize, usecols=usecols):\n            chunk = reduce_memory_usage(chunk)\n            chunk_list.append(chunk)\n        df = pd.concat(chunk_list, ignore_index=True)\n        return df\n    except Exception as e:\n        print(f\"Error loading {file_path} in chunks: {e}\")\n        return None\n\n# Step 4: Load Parquet files with memory optimization\ndef load_parquet_file(file_path, columns=None):\n    try:\n        df = pd.read_parquet(file_path, columns=columns)\n        df = reduce_memory_usage(df)\n        return df\n    except Exception as e:\n        print(f\"Error loading {file_path}: {e}\")\n        return None\n\n# Step 5: Load Data\nseries_train = load_parquet_file(series_train_path, columns=['id', 'step', 'X', 'Y', 'Z', 'enmo', 'anglez'])\nseries_test = load_parquet_file(series_test_path, columns=['id', 'step', 'X', 'Y', 'Z', 'enmo', 'anglez'])\n\ntrain_data = load_csv_in_chunks(train_file_path, usecols=['id', 'target_column_name'])  # Update 'target_column_name' as needed\ntest_data = load_csv_in_chunks(test_file_path)\ndata_dictionary = load_csv_in_chunks(data_dictionary_path)\nsample_submission = load_csv_in_chunks(sample_submission_path)\n\n# Step 6: Display data samples with conditionals\ndef display_head(df, name):\n    if df is not None:\n        print(f\"\\n{name}:\")\n        print(df.head())\n    else:\n        print(f\"{name} could not be loaded.\")\n\ndisplay_head(series_train, \"Series Train Data\")\ndisplay_head(series_test, \"Series Test Data\")\ndisplay_head(train_data, \"Training Data\")\ndisplay_head(test_data, \"Test Data\")\ndisplay_head(data_dictionary, \"Data Dictionary\")\ndisplay_head(sample_submission, \"Sample Submission\")\n\n# Step 7: Explore Series Train Data if loaded successfully\nif series_train is not None:\n    try:\n        unique_ids = series_train['id'].unique()\n        print(f\"\\nNumber of unique patient IDs in training series: {len(unique_ids)}\")\n        print(f\"Sample data for patient ID {unique_ids[0]}:\\n\")\n        print(series_train[series_train['id'] == unique_ids[0]].head())\n    except Exception as e:\n        print(f\"Error exploring series_train: {e}\")\nelse:\n    print(\"Series train data is not available for exploration.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-23T19:54:18.64223Z","iopub.execute_input":"2024-10-23T19:54:18.642961Z","iopub.status.idle":"2024-10-23T19:54:48.888691Z","shell.execute_reply.started":"2024-10-23T19:54:18.642909Z","shell.execute_reply":"2024-10-23T19:54:48.887549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}