{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport random\n\n# Load the original CSV file into a DataFrame\nfile_path = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\ndata = pd.read_csv(file_path)\n\n# Sample 100 rows randomly\nsampled_data = data.sample(n=100, random_state=42)  # Change the random_state as needed for reproducibility\n\n# Save the sampled data to test1.csv\nsampled_data.to_csv('test1.csv', index=False)\n\n# Remove the sampled rows from the original data\nremaining_data = data.drop(sampled_data.index)\n\n# Save the remaining data to a new CSV file\nremaining_data.to_csv('remaining_data.csv', index=False)\n\nprint(\"Done\")\n","metadata":{"execution":{"iopub.status.busy":"2024-03-04T08:57:09.368092Z","iopub.execute_input":"2024-03-04T08:57:09.368614Z","iopub.status.idle":"2024-03-04T08:57:10.640749Z","shell.execute_reply.started":"2024-03-04T08:57:09.368581Z","shell.execute_reply":"2024-03-04T08:57:10.639356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Load the training dataset\ntrain_data = pd.read_csv(\"remaining_data.csv\")\n\n# Check for missing values\nmissing_values = train_data.isnull().sum()\nprint(\"Missing values before handling:\")\nprint(missing_values)\n\n# Handle missing values by imputation (replace missing values with mean)\ntrain_data.fillna(train_data.mean(), inplace=True)\n\n# Alternatively, remove samples with missing values if they are too sparse\n# train_data.dropna(inplace=True)\n\n# Check again for missing values after handling\nmissing_values_after = train_data.isnull().sum()\nprint(\"\\nMissing values after handling:\")\nprint(missing_values_after)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-04T08:57:18.137065Z","iopub.execute_input":"2024-03-04T08:57:18.137521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#1\nimport pandas as pd\nimport os\nimport pyarrow.parquet as pq\n\n# Load the CSV file into a DataFrame\ncsv_file_path = ''\ndf_csv = pd.read_csv(csv_file_path)\n\n# Directory containing the .parquet files\nparquet_dir = 'D:\\Abhinaya_BTech_Notes\\OneDrive\\semesters\\SEM 6\\OPEN LAB\\train_eegs'\n\n# Iterate over the .parquet files\nfor parquet_file in os.listdir(parquet_dir):\n    if parquet_file.endswith('.parquet'):\n        parquet_path = os.path.join(parquet_dir, parquet_file)\n        # Load the .parquet file into a DataFrame\n        df_parquet = pq.read_table(parquet_path).to_pandas()\n        # Check which samples from the CSV file are in the .parquet file\n        common_samples = pd.merge(df_csv, df_parquet, how='inner', on='sample_column_name')\n        print(f\"Common samples between {csv_file_path} and {parquet_path}: {len(common_samples)}\")\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\n\nparquet_dir = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs'\ndfs = []\nfor filename in os.listdir(parquet_dir):\n    if filename.endswith(\".parquet\"):\n        eeg_id = filename.split('.')[0]\n        df = pd.read_parquet(os.path.join(parquet_dir, filename))\n        df.insert(0, 'eeg_id', eeg_id)\n        dfs.append(df)\nresult_df = pd.concat(dfs, ignore_index=True)\nresult_df.to_csv('concatenated_data.csv', index=False)\nprint(\"CSV file saved successfully.\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}