{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"},{"sourceId":8479599,"sourceType":"datasetVersion","datasetId":4517815},{"sourceId":8723242,"sourceType":"datasetVersion","datasetId":5234797},{"sourceId":8792952,"sourceType":"datasetVersion","datasetId":5286840},{"sourceId":174185912,"sourceType":"kernelVersion"},{"sourceId":177382875,"sourceType":"kernelVersion"},{"sourceId":177932770,"sourceType":"kernelVersion"},{"sourceId":185015269,"sourceType":"kernelVersion"}],"dockerImageVersionId":30664,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport glob\nfrom tqdm import tqdm\nimport gc\nfrom collections import defaultdict\nimport datetime\nimport json\nimport itertools\nimport matplotlib.pyplot as plt\nimport copy\nimport math\nfrom collections import Counter\nimport numpy as np\nimport gc\nimport time\nimport random\n\nimport whoosh_utils","metadata":{"execution":{"iopub.status.busy":"2024-07-06T05:34:45.146194Z","iopub.execute_input":"2024-07-06T05:34:45.146637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_NUM = 1000 # 10000","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# word_to_number = dict()\nword_to_pub_set = defaultdict(set)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pub_to_number = dict()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(10):\n    path = f'/kaggle/input/uspto-patent-all-claims/patent_all_{i}_claims.parquet'\n    df = pd.read_parquet(path)\n    print('len(df)', len(df))\n    \n    remove_word_set = set()\n    \n    for pub, claims in tqdm(df[['publication_number', 'claims']].values):\n        pub_to_number[pub] = len(pub_to_number)\n        pub = pub_to_number[pub]\n        \n        for word in claims.split():\n            word = 'clm:' + word\n            \n            if word in remove_word_set:\n                continue\n            \n            word_to_pub_set[word].add(pub)\n            \n            # 個数が多すぎるものは使わない\n            if len(word_to_pub_set[word]) == MAX_NUM:\n                word_to_pub_set.pop(word)\n                remove_word_set.add(word)\n        \n    del df\n    gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(word_to_pub_set), len(remove_word_set)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_to_number = dict()\nnumber_to_word = dict()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"use_words = sorted(word_to_pub_set.keys())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = '/kaggle/input/precompute-cpc-title-desc-max-10000-word-to-int/'\nword_to_number_base = pickle.load(open(base_path + 'word_to_number.pkl', 'rb'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, word in enumerate(use_words):\n    i = len(word_to_number_base) + i # 後で、base_path内のword_to_numberと結合するため\n    word_to_number[word] = i\n    number_to_word[i] = word","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min(number_to_word.keys()), max(number_to_word.keys())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_to_pub_set_new = defaultdict(set)\n\nfor word, pub_set in word_to_pub_set.items():\n    word_to_pub_set_new[word_to_number[word]] = pub_set","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_to_pub_set = word_to_pub_set_new","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_to_pubcount = defaultdict(int)\nfor word, pub_set in word_to_pub_set.items():\n    word_to_pubcount[word] = len(pub_set)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"publication_to_word = defaultdict(list)\nfor word, pub_set in tqdm(word_to_pub_set.items()):\n    for pub in pub_set:\n        publication_to_word[pub].append(word)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pickle.dump(word_to_number, open(f'word_to_number.pkl', 'wb'))\npickle.dump(number_to_word, open(f'number_to_word.pkl', 'wb'))\npickle.dump(word_to_pub_set, open(f'word_to_pub_set.pkl', 'wb'))\npickle.dump(word_to_pubcount, open(f'word_to_pubcount.pkl', 'wb'))\npickle.dump(publication_to_word, open(f'publication_to_word.pkl', 'wb'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pickle.dump(pub_to_number, open(f'pub_to_number.pkl', 'wb'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}