{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"},{"sourceId":8479599,"sourceType":"datasetVersion","datasetId":4517815},{"sourceId":9009613,"sourceType":"datasetVersion","datasetId":5428152},{"sourceId":174185912,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"markdown","source":"    The general idea is to load all infos about single word / cpc occurances from an sqlite databank and then create a query based on these.\n\n    We processed all publications to only include a single copy for each word to save space and will only be using a single or chain. \n    \n    OR will be the only used operator (also no ommited AND, the magic for this competition)\n    \n    We load the neighbors of some neighbors to catch more potential false positives.","metadata":{}},{"cell_type":"markdown","source":"## Settings","metadata":{}},{"cell_type":"code","source":"# General Settings\nTEST = True          # Either run on test or a fraction of train\nif TEST:\n    SUBSET_SIZE = 1000000 # This just means to run on all (break limit)\n    NEIGHBOR_NAME = 'target_'\nelse:\n    SUBSET_SIZE = 100\n    NEIGHBOR_NAME = 'neighbor_'\n    \nNEIGHBOR_SELECTION = [1,2,3,4,5,6,7,8,9] # Refer to index in df so [0] would be self [1] neighbor_0/target_0","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:12:02.891786Z","iopub.execute_input":"2024-07-25T13:12:02.892251Z","iopub.status.idle":"2024-07-25T13:12:02.907453Z","shell.execute_reply.started":"2024-07-25T13:12:02.892211Z","shell.execute_reply":"2024-07-25T13:12:02.906113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport sqlite3\nimport polars as pl\nfrom collections import defaultdict\nimport whoosh_utils\nimport random\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:12:02.909023Z","iopub.execute_input":"2024-07-25T13:12:02.909449Z","iopub.status.idle":"2024-07-25T13:12:40.602142Z","shell.execute_reply.started":"2024-07-25T13:12:02.909403Z","shell.execute_reply":"2024-07-25T13:12:40.600839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Test / Train","metadata":{}},{"cell_type":"code","source":"data_dir = \"/kaggle/input/uspto-explainable-ai/\"","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:12:40.603788Z","iopub.execute_input":"2024-07-25T13:12:40.604399Z","iopub.status.idle":"2024-07-25T13:12:40.610689Z","shell.execute_reply.started":"2024-07-25T13:12:40.604361Z","shell.execute_reply":"2024-07-25T13:12:40.609076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TEST:\n    test = pd.read_csv(data_dir + \"test.csv\")\nelse:\n    num_patents = SUBSET_SIZE  # WARNING number of patents in index will be 50 * num_patents\n    seed = 1\n\n    # Select num_patents publication numbers from 1975 or later\n    p_meta = pl.read_parquet(\n        os.path.join(data_dir, \"patent_metadata.parquet\"),\n        columns=[\"publication_number\", \"publication_date\"]\n    )\n    p_meta = p_meta.filter(pl.col(\"publication_date\") >= pl.date(1975, 1, 1))\n\n    # Set seed for reproducibility and sample num_patents\n    random.seed(seed)\n    train_samples = p_meta.sample(n=num_patents, with_replacement=False).select(\"publication_number\")\n\n    # Load relevant neighbors data based on selected patents\n    neighbors = pl.scan_csv(os.path.join(data_dir, \"nearest_neighbors.csv\"))\n    train_samples_set = set(train_samples.to_series().to_list())\n    neighbors_filtered = neighbors.filter(pl.col(\"publication_number\").is_in(train_samples_set))\n\n    # Collect filtered neighbors to pandas DataFrame if necessary\n    test = neighbors_filtered.collect().to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:12:40.614685Z","iopub.execute_input":"2024-07-25T13:12:40.615177Z","iopub.status.idle":"2024-07-25T13:12:40.642579Z","shell.execute_reply.started":"2024-07-25T13:12:40.615142Z","shell.execute_reply":"2024-07-25T13:12:40.640899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:12:40.644293Z","iopub.execute_input":"2024-07-25T13:12:40.644782Z","iopub.status.idle":"2024-07-25T13:12:40.680324Z","shell.execute_reply.started":"2024-07-25T13:12:40.644732Z","shell.execute_reply":"2024-07-25T13:12:40.678777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## DB Load\n    We load all data of the 2500 patents and 125000 neighbors into a runtime dictionary for O(1) calls.\n    The neighbors neighbors we just use for word counting.","metadata":{}},{"cell_type":"code","source":"def search_publication_numbers(conn, publication_numbers):\n    # Format the query to use SQL's IN clause\n    placeholders = ','.join('?' for _ in publication_numbers)\n    query = f\"SELECT * FROM info WHERE publication_number IN ({placeholders})\"\n    return pd.read_sql_query(query, conn, params=publication_numbers)\n\n\ndef process_publications(publications_orig, publications_add, batch_size=100, db_path='various_saves/all_info_desc.db'):\n    conn = sqlite3.connect(db_path)\n    runtime_dict = {}\n    feature_counts = defaultdict(int)\n\n    # Orig publications are counted and add\n    publications_orig = list(publications_orig)\n    for i in range(0, len(publications_orig), batch_size):\n        batch = publications_orig[i:i + batch_size]\n        results = search_publication_numbers(conn, batch)\n        for pub in batch:\n            all_info = results[results['publication_number'] == pub]\n            prefix = {'title': 'ti', 'abstract': 'ab', 'claims': 'clm', 'description': 'detd', 'cpc_codes': 'cpc'}  \n            all_features = set()\n            # Go through categories\n            for col in ['title', 'abstract', 'claims', 'description', 'cpc_codes']:\n                col_info = all_info[col].values[0].split()\n                # Specific update to the column\n                for feature in col_info:\n                    prefixed_feature = prefix[col] + \":\" + feature\n                    all_features.add(prefixed_feature)\n                    # Count the feature\n                    feature_counts[prefixed_feature] += 1\n            runtime_dict[pub] = all_features\n        if i % 100 == 0:\n            print(f\"Processed {i + len(batch)} publications\")\n    print(\"Processed all original publications\")\n\n    # Add publications are only counted\n    publications_add = list(publications_add)\n    for i in range(0, len(publications_add), batch_size):\n        batch = publications_add[i:i + batch_size]\n        results = search_publication_numbers(conn, batch)\n        for pub in batch:\n            all_info = results[results['publication_number'] == pub]\n            prefix = {'title': 'ti', 'abstract': 'ab', 'claims': 'clm', 'description': 'detd', 'cpc_codes': 'cpc'}  \n            all_features = set()\n            # Go through categories\n            for col in ['title', 'abstract', 'claims', 'description', 'cpc_codes']:\n                col_info = all_info[col].values[0].split()\n                # Specific update to the column\n                for feature in col_info:\n                    prefixed_feature = prefix[col] + \":\" + feature\n                    all_features.add(prefixed_feature)\n                    # Count the feature\n                    feature_counts[prefixed_feature] += 1\n        if i % 100 == 0:\n            print(f\"Processed {i + len(batch)} publications\")\n        \n    return runtime_dict, feature_counts","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:12:40.682546Z","iopub.execute_input":"2024-07-25T13:12:40.683050Z","iopub.status.idle":"2024-07-25T13:12:40.700281Z","shell.execute_reply.started":"2024-07-25T13:12:40.682995Z","shell.execute_reply":"2024-07-25T13:12:40.698761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All pub_nums for self and neighbors\nlist_pub_orig = test[\"publication_number\"].unique()\nlist_pub_orig = np.append(list_pub_orig, test[[NEIGHBOR_NAME + str(n) for n in range(50)]])\n\nset_pub_orig = set(list_pub_orig)\nprint(f\"Number of unique publication numbers: {len(set_pub_orig)}\")","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:12:40.701936Z","iopub.execute_input":"2024-07-25T13:12:40.702383Z","iopub.status.idle":"2024-07-25T13:12:40.721633Z","shell.execute_reply.started":"2024-07-25T13:12:40.702350Z","shell.execute_reply":"2024-07-25T13:12:40.719959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Add Neighbor's Neighbors**","metadata":{}},{"cell_type":"code","source":"new_additions = []\nfor i, row in test.iterrows():\n    # CARE! Overwritting far_neigbor here as experiment\n    for neighbor in NEIGHBOR_SELECTION:\n        new_additions.append(row[neighbor])    \n\n# load the relevent neighbors data based on selected patents\nneighbors = pl.read_csv(os.path.join(data_dir, \"nearest_neighbors.csv\"))\nneighbors = neighbors.filter(pl.col(\"publication_number\").is_in(new_additions))\ntest_addition = neighbors.to_pandas()\n\n# All docs\nlist_pub_addition = test_addition[\"publication_number\"].unique()\nfor n in range(50):\n    list_pub_addition = np.append(list_pub_addition, test_addition[\"neighbor_\" + str(n)])\n\nset_pub_add = set(list_pub_addition)\nset_pub_add = set_pub_add - set_pub_orig\nprint(f\"Number of unique publication numbers: {len(set_pub_add)}\")","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:12:40.723485Z","iopub.execute_input":"2024-07-25T13:12:40.723882Z","iopub.status.idle":"2024-07-25T13:13:20.863877Z","shell.execute_reply.started":"2024-07-25T13:12:40.723841Z","shell.execute_reply":"2024-07-25T13:13:20.862368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load runtime dicts and feature counts\nruntime_dict, feature_counts = process_publications(set_pub_orig, set_pub_add, batch_size=100, db_path='/kaggle/input/uspto-full-error-corrected/all_info_desc.db')","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:13:20.865503Z","iopub.execute_input":"2024-07-25T13:13:20.865970Z","iopub.status.idle":"2024-07-25T13:13:33.507234Z","shell.execute_reply.started":"2024-07-25T13:13:20.865926Z","shell.execute_reply":"2024-07-25T13:13:33.505730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Functions","metadata":{}},{"cell_type":"code","source":"def ap50est(tp, fp):\n    precisions = list()\n    n_found = 0\n    for i in range(1,51):\n        if i <= tp:\n            n_found += tp/(tp+fp)\n        precisions.append(n_found/i) \n    return sum(precisions)/50","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:13:33.508996Z","iopub.execute_input":"2024-07-25T13:13:33.509422Z","iopub.status.idle":"2024-07-25T13:13:33.517765Z","shell.execute_reply.started":"2024-07-25T13:13:33.509387Z","shell.execute_reply":"2024-07-25T13:13:33.515970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_features_tp(pubs):\n    # Construction of features for the row\n    all_features = defaultdict(int)\n    for tp in pubs:\n        for feature in runtime_dict[tp]:\n            all_features[feature] += 1\n\n    rows = []\n    for tp in pubs:\n        row = {'publication_number': tp}                  \n        # Go through categories\n        for feature in runtime_dict[tp]:\n            if feature in all_features:\n                row[feature] = 1\n        rows.append(row)\n    \n    df = pd.DataFrame(rows)\n    df.fillna(0, inplace=True)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:13:33.520013Z","iopub.execute_input":"2024-07-25T13:13:33.520523Z","iopub.status.idle":"2024-07-25T13:13:33.534891Z","shell.execute_reply.started":"2024-07-25T13:13:33.520486Z","shell.execute_reply":"2024-07-25T13:13:33.533371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example\nexample_pub = test.iloc[0]\npub_tp = [example_pub[NEIGHBOR_NAME + str(n)] for n in range(50)]\ndf_features = generate_features_tp(pub_tp)\nprint(df_features.shape)\ndf_features.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:13:33.536747Z","iopub.execute_input":"2024-07-25T13:13:33.537269Z","iopub.status.idle":"2024-07-25T13:13:34.628433Z","shell.execute_reply.started":"2024-07-25T13:13:33.537221Z","shell.execute_reply":"2024-07-25T13:13:34.627251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def df_tp_opt(df, budget=25, fp_multiplier=10.0):\n\n    estimated_score_global = 0\n    best_features_global = 0\n    tp_real_global = 0\n    fp_est_global = 0\n    \n    # Iterate to find best fp_multiplier to calculate next feature in greedy search, stop at best toal score\n    while True:\n        best_features = []\n        tp_real = []\n        fp_est = []\n\n        iter = 0\n        df_base = df.copy()\n        estimated_score_last = 0\n        \n        # Add features until score decays or\n        while df_base.shape[0] > 0:\n            iter += 1\n            # Create a df with the number of true positives for each numerical feature\n            row_tp = df_base[df_base.columns[1:]].sum(axis=0)\n            row_fp = [feature_counts[feature] - row_tp[feature] for feature in row_tp.index]\n            df_summary = pd.DataFrame({'feature': row_tp.index, 'true_positives': row_tp.values, 'false_positives': row_fp})\n\n            # Change the score to adjust for ap_est \n            df_summary['score'] = df_summary['true_positives'] / (df_summary['false_positives'] * fp_multiplier + 1) #** fp_severity_mod\n            df_summary = df_summary.sort_values(by='score', ascending=False)\n\n            # Best feature\n            best_feature = df_summary.iloc[0]['feature']\n            fp_round = df_summary.iloc[0]['false_positives']\n            tp_round = df_summary.iloc[0]['true_positives']\n\n            # Break once it gets worse\n            estimated_score = ap50est(sum(tp_real + [tp_round]), sum(fp_est + [fp_round]))\n            if estimated_score <= estimated_score_last:\n                break\n            else:\n                estimated_score_last = estimated_score\n\n                # Remove the entries where the best feature is 1\n                df_base = df_base[df_base[best_feature] == 0]\n                # Remove all features that have no true positives anymore (sum == 0)\n                df_base = df_base.loc[:, (df_base != 0).any(axis=0)]\n\n                # Add the best feature to the best features list\n                best_features.append(best_feature)\n                fp_est.append(fp_round)\n                tp_real.append(tp_round)\n\n            if iter == budget:\n                break\n        \n        if estimated_score_last > estimated_score_global:\n            estimated_score_global = estimated_score_last\n            best_features_global = best_features\n            tp_real_global = tp_real\n            fp_est_global = fp_est\n            fp_multiplier /= 1.3\n        else:\n            break\n\n    return best_features_global, tp_real_global, fp_est_global, estimated_score_global","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:13:34.633210Z","iopub.execute_input":"2024-07-25T13:13:34.633631Z","iopub.status.idle":"2024-07-25T13:13:34.649817Z","shell.execute_reply.started":"2024-07-25T13:13:34.633593Z","shell.execute_reply":"2024-07-25T13:13:34.648230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create or chain with best_features\ndef or_chain_from_features(best_features):\n    return \"(\" + \" OR \".join(best_features) + \")\"","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:13:34.651911Z","iopub.execute_input":"2024-07-25T13:13:34.653301Z","iopub.status.idle":"2024-07-25T13:13:34.670004Z","shell.execute_reply.started":"2024-07-25T13:13:34.653239Z","shell.execute_reply":"2024-07-25T13:13:34.668220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"queries = []\nsolution_rows = []\nfor i, row in test.iterrows():\n    # Add all pub num from the 50 neighbors\n    pub_tp = [row[NEIGHBOR_NAME + str(n)] for n in range(50)]\n    \n    # Generate features\n    df = generate_features_tp(pub_tp)\n\n    # Optimize\n    best_features, tp_real, fp_upper_bounds, est_score = df_tp_opt(df, budget=24, fp_multiplier=10)\n\n    # Create or chain\n    or_chain = or_chain_from_features(best_features)\n\n    # Count tokens\n    tokens_count = whoosh_utils.count_query_tokens(or_chain)\n    fp_est = sum(fp_upper_bounds)\n    tp_real = sum(tp_real)\n    solution_row = {'publication_number': row['publication_number'], 'or_chain': or_chain, 'tokens_count': tokens_count, 'fp_est': fp_est, 'tp': tp_real, 'estimated_score': est_score}\n    solution_rows.append(solution_row)\n    \n    queries.append(or_chain)\n    \n    print(i)\n\nsolution_df = pd.DataFrame(solution_rows)","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:13:34.671773Z","iopub.execute_input":"2024-07-25T13:13:34.672437Z","iopub.status.idle":"2024-07-25T13:14:10.231627Z","shell.execute_reply.started":"2024-07-25T13:13:34.672379Z","shell.execute_reply":"2024-07-25T13:14:10.226360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution_df","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:14:10.235501Z","iopub.execute_input":"2024-07-25T13:14:10.236700Z","iopub.status.idle":"2024-07-25T13:14:10.300723Z","shell.execute_reply.started":"2024-07-25T13:14:10.236560Z","shell.execute_reply":"2024-07-25T13:14:10.297193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"for query in queries:\n    tokens_used = whoosh_utils.count_query_tokens(query)\n    validator = whoosh_utils.QueryValidator().validate_query(query)\n    #print(tokens_used)","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:14:10.304809Z","iopub.execute_input":"2024-07-25T13:14:10.305843Z","iopub.status.idle":"2024-07-25T13:14:10.344195Z","shell.execute_reply.started":"2024-07-25T13:14:10.305806Z","shell.execute_reply":"2024-07-25T13:14:10.338601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign queries to submission\nsub = pd.read_csv(\"/kaggle/input/uspto-explainable-ai/sample_submission.csv\")\nsub['query'] = queries","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:14:10.351752Z","iopub.execute_input":"2024-07-25T13:14:10.355345Z","iopub.status.idle":"2024-07-25T13:14:10.380202Z","shell.execute_reply.started":"2024-07-25T13:14:10.355211Z","shell.execute_reply":"2024-07-25T13:14:10.375598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:14:10.383224Z","iopub.execute_input":"2024-07-25T13:14:10.384924Z","iopub.status.idle":"2024-07-25T13:14:10.411490Z","shell.execute_reply.started":"2024-07-25T13:14:10.384780Z","shell.execute_reply":"2024-07-25T13:14:10.407708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub","metadata":{"execution":{"iopub.status.busy":"2024-07-25T13:14:10.416379Z","iopub.execute_input":"2024-07-25T13:14:10.417564Z","iopub.status.idle":"2024-07-25T13:14:10.451432Z","shell.execute_reply.started":"2024-07-25T13:14:10.417446Z","shell.execute_reply":"2024-07-25T13:14:10.448072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}