{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"},{"sourceId":8479599,"sourceType":"datasetVersion","datasetId":4517815},{"sourceId":8723242,"sourceType":"datasetVersion","datasetId":5234797},{"sourceId":8792952,"sourceType":"datasetVersion","datasetId":5286840},{"sourceId":174185912,"sourceType":"kernelVersion"},{"sourceId":177382875,"sourceType":"kernelVersion"},{"sourceId":187069046,"sourceType":"kernelVersion"},{"sourceId":188180843,"sourceType":"kernelVersion"},{"sourceId":189012948,"sourceType":"kernelVersion"},{"sourceId":189013978,"sourceType":"kernelVersion"},{"sourceId":189014459,"sourceType":"kernelVersion"},{"sourceId":189014799,"sourceType":"kernelVersion"},{"sourceId":189019084,"sourceType":"kernelVersion"},{"sourceId":189021148,"sourceType":"kernelVersion"},{"sourceId":189238016,"sourceType":"kernelVersion"}],"dockerImageVersionId":30666,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport glob\nfrom tqdm import tqdm\nimport gc\nfrom collections import defaultdict\nimport datetime\nimport json\nimport itertools\nimport matplotlib.pyplot as plt\nimport copy\nimport math\nfrom collections import Counter\nimport numpy as np\nimport gc\nimport time\nimport random\nimport pickle\n\nimport whoosh_utils","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 実験内容  \nこれに変更  \nhttps://www.kaggle.com/code/tanakar/precompute-cpc-title-abst-claim-max-100000/output?select=word_to_pub_set.pkl  \ndescriptionをmax=100000に変更  \ndescriptionのmergeだけやる。titleとかとのmergeは後で。","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%python\n\nimport pandas as pd\nimport glob\nfrom tqdm import tqdm\nimport gc\nfrom collections import defaultdict\nimport datetime\nimport json\nimport itertools\nimport matplotlib.pyplot as plt\nimport copy\nimport math\nfrom collections import Counter\nimport numpy as np\nimport gc\nimport time\nimport random\nimport pickle\nimport os\n\n\nword_to_pub_set = defaultdict(list)\nremove_word_set = set()\n\npaths = ['/kaggle/input/preprocess-description-tpu-year-1975-max-100000/remove_word_set.pkl',\n         '/kaggle/input/preprocess-description-tpu-chunk1-max-100000/remove_word_set.pkl',\n         '/kaggle/input/preprocess-description-tpu-chunk2-max-100000/remove_word_set.pkl',\n         '/kaggle/input/preprocess-description-tpu-chunk3-max-100000/remove_word_set.pkl',\n         '/kaggle/input/preprocess-description-tpu-chunk4-1-2-max-100000/remove_word_set.pkl',\n         '/kaggle/input/preprocess-description-tpu-chunk4-2-2-max-100000/remove_word_set.pkl']\nfor path in paths:\n    _remove_word_set = pickle.load(open(path, 'rb'))\n    remove_word_set |= _remove_word_set\n\nMAX_NUM = 100000\n\npaths = ['/kaggle/input/preprocess-description-tpu-year-1975-max-100000/word_to_pub_set.pkl',\n         '/kaggle/input/preprocess-description-tpu-chunk1-max-100000/word_to_pub_set.pkl',\n         '/kaggle/input/preprocess-description-tpu-chunk2-max-100000/word_to_pub_set.pkl',\n         '/kaggle/input/preprocess-description-tpu-chunk3-max-100000/word_to_pub_set.pkl',\n         '/kaggle/input/preprocess-description-tpu-chunk4-1-2-max-100000/word_to_pub_set.pkl',\n         '/kaggle/input/preprocess-description-tpu-chunk4-2-2-max-100000/word_to_pub_set.pkl']\n\nfor path in paths:\n    _word_to_pub_set = pickle.load(open(path, 'rb'))\n    \n    for word, pub_set in tqdm(_word_to_pub_set.items()):\n        if word in remove_word_set:\n            continue\n        \n        word_to_pub_set[word].extend(pub_set)\n        \n        if len(word_to_pub_set[word]) > MAX_NUM:\n            word_to_pub_set.pop(word)\n            remove_word_set.add(word)\n            \nprint(len(word_to_pub_set), len(remove_word_set))\n\n# 重複削除\nseen = set()\nremove_words = []\nfor word, pub_set in tqdm(word_to_pub_set.items()):\n    key = tuple(sorted(pub_set))\n    \n    if key in seen:\n        remove_words.append(word)\n    else:\n        seen.add(key)\n\nprint('len(remove_words)', len(remove_words))\nfor word in remove_words:\n    word_to_pub_set.pop(word)\nprint(len(word_to_pub_set))\n\npickle.dump(word_to_pub_set, open(f'word_to_pub_set_desc_only.pkl', 'wb'))","metadata":{"execution":{"iopub.status.busy":"2024-07-09T06:00:07.18771Z","iopub.execute_input":"2024-07-09T06:00:07.18815Z","iopub.status.idle":"2024-07-09T06:08:43.344698Z","shell.execute_reply.started":"2024-07-09T06:00:07.188123Z","shell.execute_reply":"2024-07-09T06:08:43.342951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%python\n\nimport pandas as pd\nimport glob\nfrom tqdm import tqdm\nimport gc\nfrom collections import defaultdict\nimport datetime\nimport json\nimport itertools\nimport matplotlib.pyplot as plt\nimport copy\nimport math\nfrom collections import Counter\nimport numpy as np\nimport gc\nimport time\nimport random\nimport pickle\nimport os\n\npath = f'word_to_pub_set_desc_only.pkl'\nword_to_pub_set = pickle.load(open(path, 'rb'))\n\nword_to_number = dict()\nnumber_to_word = dict()\n\nuse_words = sorted(word_to_pub_set.keys())\n\n# base_path = '/kaggle/input/precompute-claim-set-max-10000-add-claims-lowmem/'\n# base_path = '/kaggle/input/precompute-claim-set-max-10000-add-claims-allyear/'\n# base_path = '/kaggle/input/precompute-claim-set-max-10000-add-claims-fix/'\n# base_path = '/kaggle/input/precompute-cpc-title-desc-max-100000-allyear-claim/'\n# base_path = '/kaggle/input/precompute-cpc-title-abst-claim-max-100000/'\nbase_path = '/kaggle/input/precompute-cpc-title-abst-claim-max-inf/'\nword_to_number_base = pickle.load(open(base_path + 'word_to_number.pkl', 'rb'))\n\nfor i, word in enumerate(use_words):\n    i = len(word_to_number_base) + i # 後で、base_path内のword_to_numberと結合するため\n    word_to_number[word] = i\n    number_to_word[i] = word\n\nmin(number_to_word.keys()), max(number_to_word.keys())\n\nword_to_pub_set_new = defaultdict(list)\nfor word, pub_set in word_to_pub_set.items():\n    word_to_pub_set_new[word_to_number[word]] = pub_set\nword_to_pub_set = word_to_pub_set_new\n\nword_to_pubcount = defaultdict(int)\nfor word, pub_set in word_to_pub_set.items():\n    word_to_pubcount[word] = len(pub_set)\n    \npublication_to_word = defaultdict(list)\nfor word, pub_set in tqdm(word_to_pub_set.items()):\n    for pub in pub_set:\n        publication_to_word[pub].append(word)\n    \n\nprint('len(publication_to_word)', len(publication_to_word))\n        \noutput_dir = 'desc/'\nos.makedirs(output_dir, exist_ok=True)\n\npickle.dump(word_to_number, open(output_dir+f'word_to_number.pkl', 'wb'))\npickle.dump(number_to_word, open(output_dir+f'number_to_word.pkl', 'wb'))\npickle.dump(word_to_pub_set, open(output_dir+f'word_to_pub_set.pkl', 'wb'))\npickle.dump(word_to_pubcount, open(output_dir+f'word_to_pubcount.pkl', 'wb'))\npickle.dump(publication_to_word, open(output_dir+f'publication_to_word.pkl', 'wb'))","metadata":{"execution":{"iopub.status.busy":"2024-07-09T06:08:43.347097Z","iopub.execute_input":"2024-07-09T06:08:43.348209Z","iopub.status.idle":"2024-07-09T06:13:37.625486Z","shell.execute_reply.started":"2024-07-09T06:08:43.348149Z","shell.execute_reply":"2024-07-09T06:13:37.62426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}