{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install rdkit-pypi  ","metadata":{"execution":{"iopub.status.busy":"2024-06-22T09:22:51.569590Z","iopub.execute_input":"2024-06-22T09:22:51.569983Z","iopub.status.idle":"2024-06-22T09:23:05.343077Z","shell.execute_reply.started":"2024-06-22T09:22:51.569952Z","shell.execute_reply":"2024-06-22T09:23:05.341569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pyarrow.parquet as pq","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-22T09:23:41.071010Z","iopub.execute_input":"2024-06-22T09:23:41.071404Z","iopub.status.idle":"2024-06-22T09:23:41.078006Z","shell.execute_reply.started":"2024-06-22T09:23:41.071377Z","shell.execute_reply":"2024-06-22T09:23:41.076494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = '/kaggle/input/leash-BELKA/train.parquet'\n\n# Read metadata from the Parquet file\nmetadata = pq.read_metadata(file_path)\n\n# Access the number of rows from the metadata\ntotal_rows = metadata.num_rows\n\n# Print the total number of rows\nprint(f\"Total number of rows in the train.parquet: {total_rows}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-22T09:23:42.904290Z","iopub.execute_input":"2024-06-22T09:23:42.904765Z","iopub.status.idle":"2024-06-22T09:23:42.921056Z","shell.execute_reply.started":"2024-06-22T09:23:42.904720Z","shell.execute_reply":"2024-06-22T09:23:42.919817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = '/kaggle/input/leash-BELKA/train.parquet'\nsample_proportion = 1000000/total_rows\n\n# Open Parquet file reader\npq_file = pq.ParquetFile(file_path)\nnum_row_groups = pq_file.num_row_groups\n\n# Initialize an empty list to store sampled DataFrames\nsampled_dfs = []\n\n# Randomly sample from each row group\nfor i in range(num_row_groups):\n    table = pq_file.read_row_group(i)\n    df = table.to_pandas()\n    sampled_df = df.sample(n=int(len(df)* sample_proportion), random_state=42)  # Adjust sample size as needed\n    sampled_dfs.append(sampled_df)\n\n# Concatenate all sampled DataFrames\ntrain_random_subset = pd.concat(sampled_dfs)\ntrain_random_subset.to_csv('/kaggle/working/train_random_subset.csv', index=False)\ntrain_random_subset.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-22T09:23:49.217086Z","iopub.execute_input":"2024-06-22T09:23:49.217943Z","iopub.status.idle":"2024-06-22T09:27:10.955447Z","shell.execute_reply.started":"2024-06-22T09:23:49.217904Z","shell.execute_reply":"2024-06-22T09:27:10.954296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_random_subset)","metadata":{"execution":{"iopub.status.busy":"2024-06-22T09:29:38.820515Z","iopub.execute_input":"2024-06-22T09:29:38.821488Z","iopub.status.idle":"2024-06-22T09:29:38.827659Z","shell.execute_reply.started":"2024-06-22T09:29:38.821448Z","shell.execute_reply":"2024-06-22T09:29:38.826537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://www.kaggle.com/code/nitishadhikari/random-sampling-train-dataset-neurips-2024","metadata":{}}]}