{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. INTRODUCTION\n\n- read input files. but these dataset is too large in single time.so I use dask library.\n- And I check unique values of each column between train and test.","metadata":{}},{"cell_type":"markdown","source":"# 1. prepare\n\n## 1.1. import libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-01T00:03:49.866564Z","iopub.execute_input":"2024-05-01T00:03:49.866986Z","iopub.status.idle":"2024-05-01T00:03:50.808574Z","shell.execute_reply.started":"2024-05-01T00:03:49.866956Z","shell.execute_reply":"2024-05-01T00:03:50.807490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dask.dataframe as dd\nimport time","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:03:50.810514Z","iopub.execute_input":"2024-05-01T00:03:50.810916Z","iopub.status.idle":"2024-05-01T00:03:51.950963Z","shell.execute_reply.started":"2024-05-01T00:03:50.810889Z","shell.execute_reply":"2024-05-01T00:03:51.949998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2. define function","metadata":{}},{"cell_type":"code","source":"def get_max_iteration_no(val, upper_val=25): \n    return int(min(val, upper_val))\n    ","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:03:51.952157Z","iopub.execute_input":"2024-05-01T00:03:51.952625Z","iopub.status.idle":"2024-05-01T00:03:51.957625Z","shell.execute_reply.started":"2024-05-01T00:03:51.952598Z","shell.execute_reply":"2024-05-01T00:03:51.956492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3. set input and output parameteres","metadata":{}},{"cell_type":"code","source":"prms_in_ = {\n    \"train\": {\n        \"file\": {\n            \"path\": \"/kaggle/input/leash-BELKA\", \n            \"name\": \"train.parquet\"\n        }\n    }, \n    \"test\": {\n        \"file\": {\n            \"path\": \"/kaggle/input/leash-BELKA\", \n            \"name\": \"test.parquet\"\n        }\n    }\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:03:51.958674Z","iopub.execute_input":"2024-05-01T00:03:51.958981Z","iopub.status.idle":"2024-05-01T00:03:51.966729Z","shell.execute_reply.started":"2024-05-01T00:03:51.958956Z","shell.execute_reply":"2024-05-01T00:03:51.965739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prms_out_ = {\n    \"unique_value_summary\": {\n        \"file\": {\n            \"path\": \"/kaggle/working/leash-BELKA/001_unique_value_summary\", \n            \"name\": \"unique_value_summary.csv\"\n        }\n    }\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:03:51.969863Z","iopub.execute_input":"2024-05-01T00:03:51.970917Z","iopub.status.idle":"2024-05-01T00:03:51.976142Z","shell.execute_reply.started":"2024-05-01T00:03:51.970884Z","shell.execute_reply":"2024-05-01T00:03:51.975184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Read files","metadata":{}},{"cell_type":"code","source":"ddf_ = {}\nfor info_lgcl_nm in [\"train\", \"test\"]: \n    ddf_[info_lgcl_nm] = dd.read_parquet(\n        \"/\".join([v for k, v in prms_in_[info_lgcl_nm][\"file\"].items()])\n    )","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:03:51.977373Z","iopub.execute_input":"2024-05-01T00:03:51.977757Z","iopub.status.idle":"2024-05-01T00:03:52.069727Z","shell.execute_reply.started":"2024-05-01T00:03:51.977721Z","shell.execute_reply":"2024-05-01T00:03:52.068674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Count unique values","metadata":{}},{"cell_type":"code","source":"clnms_unique_check = [\"buildingblock1_smiles\", \"buildingblock2_smiles\", \"buildingblock3_smiles\", \"molecule_smiles\", \"protein_name\"]","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:03:52.070896Z","iopub.execute_input":"2024-05-01T00:03:52.071578Z","iopub.status.idle":"2024-05-01T00:03:52.075829Z","shell.execute_reply.started":"2024-05-01T00:03:52.071549Z","shell.execute_reply":"2024-05-01T00:03:52.074895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.1. count unique values in each partitions","metadata":{}},{"cell_type":"code","source":"unq_vals_raw_ = {\n    \"train\": {}, \n    \"test\": {}\n}\nfor clnm in clnms_unique_check: \n    unq_vals_raw_[\"train\"][clnm]={}\n    unq_vals_raw_[\"test\"][clnm]={}","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:03:52.077260Z","iopub.execute_input":"2024-05-01T00:03:52.077913Z","iopub.status.idle":"2024-05-01T00:03:52.084248Z","shell.execute_reply.started":"2024-05-01T00:03:52.077878Z","shell.execute_reply":"2024-05-01T00:03:52.083376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## info_lgcl_nm = \"train\"\nfor info_lgcl_nm in [\"train\", \"test\"]: \n    t0_0 = time.time()\n    \n    max_iteration = get_max_iteration_no(ddf_[info_lgcl_nm].npartitions, ddf_[info_lgcl_nm].npartitions)\n    for i1, ddf_prt in enumerate(ddf_[info_lgcl_nm].partitions): \n    ## for i1 in range(0, max_iteration, 1): \n        t1_0 = time.time()\n\n        ddf_prt = ddf_[info_lgcl_nm].get_partition(i1)\n        df_tmp = ddf_prt.compute()\n\n        for clnm in clnms_unique_check: \n            unq_vals_raw_[info_lgcl_nm][clnm][i1] = df_tmp.loc[:, clnm].unique().tolist()\n\n        t1_1 = time.time()\n        print(\"{}/{}, elapsed time: {:0.03f} sec.\".format(i1+1, max_iteration, t1_1 - t1_0))\n\n        del df_tmp\n    \n    t0_1 = time.time()\n    print(\"++ {}, elapsed time: {:0.03f} sec.\".format(info_lgcl_nm, t0_1 - t0_0))\n    print(\"\")","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-05-01T00:03:52.085512Z","iopub.execute_input":"2024-05-01T00:03:52.085872Z","iopub.status.idle":"2024-05-01T00:07:57.722338Z","shell.execute_reply.started":"2024-05-01T00:03:52.085838Z","shell.execute_reply":"2024-05-01T00:07:57.721278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.2. get union set of unique values among each partitions","metadata":{}},{"cell_type":"code","source":"unq_vals_ = {}\nfor info_lgcl_nm in [\"train\", \"test\"]: \n    t0_0 = time.time()\n    \n    unq_vals_[info_lgcl_nm] = {}\n\n    for clnm in clnms_unique_check: \n        t1_0 = time.time()\n\n        tmp_list = []\n        for k, v in unq_vals_raw_[info_lgcl_nm][clnm].items(): \n            tmp_list.extend(v)\n\n        unq_vals_[info_lgcl_nm][clnm] = pd.Series(tmp_list).drop_duplicates()\n\n        t1_1 = time.time()\n        print(\"{}, elapsed time: {:0.03f} sec.\".format(clnm, t1_1 - t1_0))\n    \n    t0_1 = time.time()\n    print(\"++ {}, elapsed time: {:0.03f} sec.\".format(info_lgcl_nm, t0_1 - t0_0))\n    print(\"\")","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:07:57.723773Z","iopub.execute_input":"2024-05-01T00:07:57.724215Z","iopub.status.idle":"2024-05-01T00:09:37.226010Z","shell.execute_reply.started":"2024-05-01T00:07:57.724178Z","shell.execute_reply":"2024-05-01T00:09:37.224877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.3. count unique values of train, test, train-test union set, intersection set.","metadata":{}},{"cell_type":"code","source":"uniq_val_summary_={}\n\n# clnm = \"buildingblock1_smiles\"\nfor clnm in clnms_unique_check: \n    t0_0 = time.time()\n\n    unq_val_merge = pd.concat(\n        [\n            unq_vals_[\"train\"][clnm], \n            unq_vals_[\"test\"][clnm]\n        ], axis=0\n    ).drop_duplicates()\n\n    cond_intersection = unq_vals_[\"test\"][clnm].isin(unq_vals_[\"train\"][clnm])\n\n    uniq_val_summary_[clnm] ={\n        \"train\": len(unq_vals_[\"train\"][clnm]), \n        \"test\": len(unq_vals_[\"test\"][clnm]), \n        \"total(union)\": len(unq_val_merge), \n        \"common(intersection)\": cond_intersection.sum()\n    }    \n\n    t0_1 = time.time()\n    print(\"++ {}, elapsed time: {:0.03f} sec.\".format(info_lgcl_nm, t0_1 - t0_0))\n    ","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:09:37.227294Z","iopub.execute_input":"2024-05-01T00:09:37.227610Z","iopub.status.idle":"2024-05-01T00:12:16.889857Z","shell.execute_reply.started":"2024-05-01T00:09:37.227584Z","shell.execute_reply":"2024-05-01T00:12:16.887889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. OUTPUT","metadata":{}},{"cell_type":"code","source":"os.makedirs(prms_out_[\"unique_value_summary\"][\"file\"][\"path\"], exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:12:16.892385Z","iopub.execute_input":"2024-05-01T00:12:16.892785Z","iopub.status.idle":"2024-05-01T00:12:16.900691Z","shell.execute_reply.started":"2024-05-01T00:12:16.892756Z","shell.execute_reply":"2024-05-01T00:12:16.899760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(uniq_val_summary_).T.to_csv(\n    \"/\".join([v for k, v in prms_out_[\"unique_value_summary\"][\"file\"].items()]), \n    sep=\"\\t\",  \n    quotechar=\"\\\"\",  \n    quoting=0, ## 0: QUOTE_MINIMAL, 1: QUOTE_ALL, 2: QUOTE_NONNUMERIC, 3: QUOTE_NONE \n    encoding=\"utf-8\", ## \"utf-8\", \"shift_jis\" \n    header=True,  \n    index=True    \n)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:12:16.901826Z","iopub.execute_input":"2024-05-01T00:12:16.902377Z","iopub.status.idle":"2024-05-01T00:12:16.924756Z","shell.execute_reply.started":"2024-05-01T00:12:16.902347Z","shell.execute_reply":"2024-05-01T00:12:16.923769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(uniq_val_summary_).T","metadata":{"execution":{"iopub.status.busy":"2024-05-01T00:12:16.928098Z","iopub.execute_input":"2024-05-01T00:12:16.928502Z","iopub.status.idle":"2024-05-01T00:12:16.947314Z","shell.execute_reply.started":"2024-05-01T00:12:16.928476Z","shell.execute_reply":"2024-05-01T00:12:16.946274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# X. ref\n\n- [dask dataframe to pandas](https://www.google.com/search?q=dask+dataframe+to+pandas&client=ubuntu&hs=jzc&sca_esv=733dc95306283186&channel=fs&ei=qxAwZvreIbrV2roPzMiHaA&oq=dask+dataframe+&gs_lp=Egxnd3Mtd2l6LXNlcnAiD2Rhc2sgZGF0YWZyYW1lICoCCAAyBRAAGIAEMgUQABiABDIFEAAYgAQyBRAAGIAEMgUQABiABDIFEAAYgAQyBRAAGIAEMgUQABiABDIFEAAYgAQyBRAAGIAESJZhUKM9WNpYcAN4AZABAJgBjQGgAfUIqgEDMS45uAEDyAEA-AEBmAINoAKOCsICChAAGLADGNYEGEfCAg0QABiABBiwAxhDGIoFwgILEAAYgAQYhgMYigXCAggQABiABBiiBMICCxAAGIAEGJECGIoFwgIGEAAYFhgemAMAiAYBkAYKkgcDNC45oAfBKg&sclient=gws-wiz-serp)\n    - [https://docs.dask.org/en/latest/generated/dask.dataframe.DataFrame.partitions.html#dask.dataframe.DataFrame.partitions](https://docs.dask.org/en/latest/generated/dask.dataframe.DataFrame.partitions.html#dask.dataframe.DataFrame.partitions)\n    - [https://www.coiled.io/blog/converting-a-dask-dataframe-to-a-pandas-dataframe](https://www.coiled.io/blog/converting-a-dask-dataframe-to-a-pandas-dataframe)\n    - [https://stackoverflow.com/questions/39008391/how-to-transform-dask-dataframe-to-pd-dataframe](https://stackoverflow.com/questions/39008391/how-to-transform-dask-dataframe-to-pd-dataframe)\n    ","metadata":{}}]}