{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":106680,"databundleVersionId":13374319,"sourceType":"competition"},{"sourceId":13671760,"sourceType":"datasetVersion","datasetId":8693105}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:51:35.479518Z","iopub.execute_input":"2025-11-09T21:51:35.480119Z","iopub.status.idle":"2025-11-09T21:51:35.484436Z","shell.execute_reply.started":"2025-11-09T21:51:35.480089Z","shell.execute_reply":"2025-11-09T21:51:35.483452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta_dataset8 = pd.read_csv(\"/kaggle/input/adaptive-immune-profiling-challenge-2025/train_datasets/train_datasets/train_dataset_8/metadata.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.572585Z","iopub.execute_input":"2025-11-09T21:28:42.572839Z","iopub.status.idle":"2025-11-09T21:28:42.584149Z","shell.execute_reply.started":"2025-11-09T21:28:42.572820Z","shell.execute_reply":"2025-11-09T21:28:42.583252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta_dataset8.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.585355Z","iopub.execute_input":"2025-11-09T21:28:42.585690Z","iopub.status.idle":"2025-11-09T21:28:42.600004Z","shell.execute_reply.started":"2025-11-09T21:28:42.585661Z","shell.execute_reply":"2025-11-09T21:28:42.599155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Research notes \n\n# Specific HLA genes may be linked to autoimmune diseases\n\n# HLA-I genes present antigens from inside the cell to cytotoxic T cells (CD8+)\n# The presented (e.g. viral) peptides then attract cells that destroy the infected cell \n\n# HLA-II genes present antigens from outside the cell to T-helper cells (CD4+)\n# These particular antigens (through a relatively lengthy process) stimulate antibody-producing B-cells to produce antibodies to that specific antigen.\n# This results in antibodies marking cells to be later on destroyed by immune system","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.600953Z","iopub.execute_input":"2025-11-09T21:28:42.601221Z","iopub.status.idle":"2025-11-09T21:28:42.617845Z","shell.execute_reply.started":"2025-11-09T21:28:42.601201Z","shell.execute_reply":"2025-11-09T21:28:42.616974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# One hot is best here due to non-ordered characteristics \ngroup_encoding = pd.get_dummies(meta_dataset8['study_group_description'], prefix=\"group\")\ngroup_encoding.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.618841Z","iopub.execute_input":"2025-11-09T21:28:42.619222Z","iopub.status.idle":"2025-11-09T21:28:42.645727Z","shell.execute_reply.started":"2025-11-09T21:28:42.619196Z","shell.execute_reply":"2025-11-09T21:28:42.644869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# One hot is best here due to non-ordered characteristics \nsex_encoding = pd.get_dummies(meta_dataset8['sex'], prefix=\"sex\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.646639Z","iopub.execute_input":"2025-11-09T21:28:42.646940Z","iopub.status.idle":"2025-11-09T21:28:42.664241Z","shell.execute_reply.started":"2025-11-09T21:28:42.646894Z","shell.execute_reply":"2025-11-09T21:28:42.663204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# We have repeatability in gene alleles, trying to extract to see if there's any correlation\n# TBD: does number matter or should we encode into smth that will not introduce bias? \n# Upon further research - was a good decision. Having e.g. A24 allele increases the chance of disease manyfold\n\ndef gene_split(column=None):\n    return column.str.split(';', expand=True)\n\ntarget_column = ['A', 'B', 'C', 'DPA1', 'DPB1',\t'DQA1',\t'DQB1', 'DRB1',\t'DRB3',\t'DRB4', 'DRB5']\n\nfor item in target_column:\n    res = gene_split(meta_dataset8[item])\n    meta_dataset8[item + 'pref'], meta_dataset8[item + 'post'] = res[0], res[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.665180Z","iopub.execute_input":"2025-11-09T21:28:42.665493Z","iopub.status.idle":"2025-11-09T21:28:42.705722Z","shell.execute_reply.started":"2025-11-09T21:28:42.665466Z","shell.execute_reply":"2025-11-09T21:28:42.704756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta_dataset8 = pd.merge(meta_dataset8, sex_encoding, left_index=True, right_index=True)\nmeta_dataset8.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.706545Z","iopub.execute_input":"2025-11-09T21:28:42.706759Z","iopub.status.idle":"2025-11-09T21:28:42.740414Z","shell.execute_reply.started":"2025-11-09T21:28:42.706741Z","shell.execute_reply":"2025-11-09T21:28:42.739616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta_dataset8 = pd.concat([meta_dataset8, group_encoding], axis=1)\nmeta_dataset8.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.741358Z","iopub.execute_input":"2025-11-09T21:28:42.741657Z","iopub.status.idle":"2025-11-09T21:28:42.761860Z","shell.execute_reply.started":"2025-11-09T21:28:42.741636Z","shell.execute_reply":"2025-11-09T21:28:42.760954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Gene allele research\n# Certain serotypes e.g. A*24:XX can influence the disease rate \n# Certain combination of serotypes e.g. A*24:XX and ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.762798Z","iopub.execute_input":"2025-11-09T21:28:42.763338Z","iopub.status.idle":"2025-11-09T21:28:42.778148Z","shell.execute_reply.started":"2025-11-09T21:28:42.763314Z","shell.execute_reply":"2025-11-09T21:28:42.777294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Are nulls in dataset equal to Null protein (dysfunctional allelle) or? \n# Will treat nan as missing data, and N/A will be encoded to 0 and treated as null protein\n\nmeta_dataset8['Bpost'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.778975Z","iopub.execute_input":"2025-11-09T21:28:42.779863Z","iopub.status.idle":"2025-11-09T21:28:42.795867Z","shell.execute_reply.started":"2025-11-09T21:28:42.779829Z","shell.execute_reply":"2025-11-09T21:28:42.794887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta_dataset8.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.796870Z","iopub.execute_input":"2025-11-09T21:28:42.797194Z","iopub.status.idle":"2025-11-09T21:28:42.816171Z","shell.execute_reply.started":"2025-11-09T21:28:42.797171Z","shell.execute_reply":"2025-11-09T21:28:42.815306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta_dataset8 = meta_dataset8.dropna()\n\nfor column in meta_dataset8.columns.tolist():\n    meta_dataset8[column] = meta_dataset8[column].replace(0, '0')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.817008Z","iopub.execute_input":"2025-11-09T21:28:42.817250Z","iopub.status.idle":"2025-11-09T21:28:42.854319Z","shell.execute_reply.started":"2025-11-09T21:28:42.817231Z","shell.execute_reply":"2025-11-09T21:28:42.853229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_column = ['A', 'B', 'C', 'DPA1', 'DPB1',\t'DQA1',\t'DQB1', 'DRB1',\t'DRB3',\t'DRB4', 'DRB5']\ntarget_column += ['study_group_description', 'sex', 'repertoire_id']\nmeta_dataset8 = meta_dataset8.drop(columns=target_column)\nmeta_dataset8.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.855289Z","iopub.execute_input":"2025-11-09T21:28:42.855518Z","iopub.status.idle":"2025-11-09T21:28:42.878770Z","shell.execute_reply.started":"2025-11-09T21:28:42.855501Z","shell.execute_reply":"2025-11-09T21:28:42.877844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"category_columns_old = ['A', 'B', 'C', 'DPA1', 'DPB1',\t'DQA1',\t'DQB1', 'DRB1',\t'DRB3',\t'DRB4', 'DRB5']\n\ncategory_columns = []\nfor item in category_columns_old:\n    category_columns.extend([item + 'pref', item + 'post'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.879732Z","iopub.execute_input":"2025-11-09T21:28:42.880011Z","iopub.status.idle":"2025-11-09T21:28:42.894213Z","shell.execute_reply.started":"2025-11-09T21:28:42.879978Z","shell.execute_reply":"2025-11-09T21:28:42.893290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(category_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.895109Z","iopub.execute_input":"2025-11-09T21:28:42.895350Z","iopub.status.idle":"2025-11-09T21:28:42.912447Z","shell.execute_reply.started":"2025-11-09T21:28:42.895332Z","shell.execute_reply":"2025-11-09T21:28:42.911440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport category_encoders as ce\n\n# Allele pairs (with my limited knowledge) are not ordinal -> we need to encode. One-hot is a bit too much, creates too many features\n# As we need to predict if target variable is true or false (i.e. do binary classification), the best here would be WoE encoding\n# That will allow model to take into account weight of evidence, or influence on specific target variable\n\n\nX_train, y_train, filename = meta_dataset8.drop(columns=['label_positive', 'filename']), meta_dataset8['label_positive'], meta_dataset8['filename']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.913257Z","iopub.execute_input":"2025-11-09T21:28:42.913465Z","iopub.status.idle":"2025-11-09T21:28:42.931280Z","shell.execute_reply.started":"2025-11-09T21:28:42.913449Z","shell.execute_reply":"2025-11-09T21:28:42.930447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"woe_encoder = ce.WOEEncoder(cols=category_columns)\nX_train_woe = woe_encoder.fit_transform(X_train[category_columns], y_train).add_suffix('_woe')\nX_train_woe.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:42.932238Z","iopub.execute_input":"2025-11-09T21:28:42.932488Z","iopub.status.idle":"2025-11-09T21:28:43.157115Z","shell.execute_reply.started":"2025-11-09T21:28:42.932466Z","shell.execute_reply":"2025-11-09T21:28:43.156184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = X_train.drop(columns=category_columns)\nX_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:43.158156Z","iopub.execute_input":"2025-11-09T21:28:43.158462Z","iopub.status.idle":"2025-11-09T21:28:43.169824Z","shell.execute_reply.started":"2025-11-09T21:28:43.158437Z","shell.execute_reply":"2025-11-09T21:28:43.168944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = pd.concat([X_train, X_train_woe], axis=1) \nX_train = pd.concat([X_train, filename], axis=1) \nX_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:43.170810Z","iopub.execute_input":"2025-11-09T21:28:43.171138Z","iopub.status.idle":"2025-11-09T21:28:43.202656Z","shell.execute_reply.started":"2025-11-09T21:28:43.171113Z","shell.execute_reply":"2025-11-09T21:28:43.201875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Upon further thinking, I've decided to drop the CTRL / SDR / T1D columns\n# They can introduce noise into the prediction as we should solely focus on the gene influence, and find relation of genes with target (y_train)\n\nX_train = X_train.drop(columns=['group_CTRL', 'group_FDR', 'group_SDR', 'group_T1D'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:43.203494Z","iopub.execute_input":"2025-11-09T21:28:43.203738Z","iopub.status.idle":"2025-11-09T21:28:43.209143Z","shell.execute_reply.started":"2025-11-09T21:28:43.203710Z","shell.execute_reply":"2025-11-09T21:28:43.208300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# TCR beta CDR3 is the highly variable region of the T-cell receptor's beta chain \n# that is critical for recognizing specific peptide-MHC complexes. \n# It is generated by the imprecise joining of gene segments (V(D)J recombination), \n# resulting in a unique sequence and length for each T-cell clone. \n\n# These aminoacids then are directly responsible to interact with the MHC + peptide complexes. \n# Therefore if we look at correlation of TCR beta CDR3 + HLA sequence,\n# we can understand if T-cell would bind to cell that exibits HLA sequence,\n# without knowing which peptide is exibited (which will introduce noise).\n\n# The sequence is the result of V(D)J recombination, \n# where a V-beta gene segment joins with a D-beta segment (in some cases) \n# and then a J-beta gene segment, with the exact length and amino acid sequence \n# being determined by the specific segments chosen and the junctional diversity \n# that occurs during the process.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:43.209978Z","iopub.execute_input":"2025-11-09T21:28:43.210294Z","iopub.status.idle":"2025-11-09T21:28:43.226969Z","shell.execute_reply.started":"2025-11-09T21:28:43.210266Z","shell.execute_reply":"2025-11-09T21:28:43.226049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Features that would be meaningful to extract:\n# V family, V allele (D and J similar, D will have a lot of NAs as they are not always used -> drop D?)\n# Length of aminoacid sequence \n# Templates \n\n# For CDR3 sequence, question is how to encode it: \n# 1. Split per region (found somewhere on internet): pos 1-3, middle, last 3-4 \n# Learned that since CDR3beta is very variable, it doesn't make sense / create valuable info\n# 2. Use atchley factors (https://www.pnas.org/doi/full/10.1073/pnas.0408677102) to encode\n# Based on research, it looks like for function of TCR complex, the physical properties are important\n# Atchley factors actually describe them\n# Decided to go with atchley factors, looks legit? (again, not an expert)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:43.227951Z","iopub.execute_input":"2025-11-09T21:28:43.228299Z","iopub.status.idle":"2025-11-09T21:28:43.241090Z","shell.execute_reply.started":"2025-11-09T21:28:43.228272Z","shell.execute_reply":"2025-11-09T21:28:43.240163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nATCHLEY_FACTORS = {\n    'A': [-0.591, -1.302, -0.733, 1.570, -0.146],\n    'C': [-1.343, 0.465, -0.862, -1.020, -0.255],\n    'D': [1.050, 0.302, -3.656, -0.259, -3.242],\n    'E': [1.357, -1.453, 1.477, 0.113, -0.837],\n    'F': [-1.006, -0.590, 1.891, -0.397, 0.412],\n    'G': [-0.384, 1.652, 1.330, 1.045, 2.064],\n    'H': [0.336, -0.417, -1.673, -1.474, -0.078],\n    'I': [-1.239, -0.547, 2.131, 0.393, 0.816],\n    'K': [1.831, -0.561, 0.533, -0.277, 1.648],\n    'L': [-1.019, -0.987, -1.505, 1.266, -0.912],\n    'M': [-0.663, -1.524, 2.219, -1.005, 1.212],\n    'N': [0.945, 0.828, 1.299, -0.169, 0.933],\n    'P': [0.189, 2.081, -1.628, 0.421, -1.392],\n    'Q': [0.931, -0.179, -3.005, -0.503, -1.853],\n    'R': [1.538, -0.055, 1.502, 0.440, 2.897],\n    'S': [-0.228, 1.399, -4.760, 0.670, -2.647],\n    'T': [-0.032, 0.326, 2.213, 0.908, 1.313],\n    'V': [-1.337, -0.279, -0.544, 1.242, -1.262],\n    'W': [-0.595, 0.009, 0.672, -2.128, -0.184],\n    'Y': [0.260, 0.830, 3.097, -0.838, 1.512],\n}\n\nATCHLEY_ARRAY = np.array([ATCHLEY_FACTORS.get(aa, [0, 0, 0, 0, 0]) for aa in 'ACDEFGHIKLMNPQRSTVWY'])\nAA_TO_INDEX = {aa: i for i, aa in enumerate('ACDEFGHIKLMNPQRSTVWY')}\n\n# AI-generated, human-optimized (otherwise would be too slow)\ndef encode_atchley_kmer(sequence, k=4, normalize=True):\n    seq_len = len(sequence)\n    n_kmers = seq_len - k + 1\n    if n_kmers <= 0:\n        return np.zeros(5)\n    kmer_features = np.zeros((n_kmers, 5))\n    valid_count = 0\n    for i in range(n_kmers):\n        kmer = sequence[i:i+k]\n        try:\n            indices = [AA_TO_INDEX[aa] for aa in kmer]\n            kmer_features[valid_count] = ATCHLEY_ARRAY[indices].mean(axis=0)\n            valid_count += 1\n        except KeyError:\n            continue\n    if valid_count == 0:\n        return np.zeros(5)\n    result = kmer_features[:valid_count].mean(axis=0)\n    if normalize:\n        norm = np.linalg.norm(result)\n        if norm > 0:\n            result /= norm\n    return result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:13:38.223916Z","iopub.execute_input":"2025-11-09T22:13:38.224242Z","iopub.status.idle":"2025-11-09T22:13:38.239806Z","shell.execute_reply.started":"2025-11-09T22:13:38.224220Z","shell.execute_reply":"2025-11-09T22:13:38.238793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def gene_split(column=None):\n    return column.str.split('-', expand=True)\n    \ndef gene_cleaning(column=None, target='j'):\n    return column.str.strip('TCRB' + target.upper())\n    \ndef process_sequencing_file(df, targets=['v', 'j', 'd']):\n    # for AtchleyKmerEncoder, you have to choose k-mer length \n    # I am going to partially copy this paper: https://pmc.ncbi.nlm.nih.gov/articles/PMC7993519/\n    # It uses one-hot encoded V/J/D genes + 4-kmer encoded Junctions\n    # I am not going to skip any AA's from start or end bc I wasn't able to find any evidence it makes sense with CDR3 beta \n    #atchley(k_mer=4, abundance=RELATIVE_ABUNDANCE, skip_first_n_aa=0,skip_last_n_aa=0, normalize_all_features=False) \n    \n\n\n    for target in targets:\n        column = target + \"_call\"\n        df[column] = gene_cleaning(df[column], target)\n        res = gene_split(df[column]) \n        # decided to ignore p3 as it is very rare\n        df[column + 'f'], df[column + 's'] = res[0], res[1]\n        df = df.drop(columns=[column])\n\n        for item in ['f', 's']:\n            df[column + item] = df[column + item].replace('unknown', '0')\n            df[column + item] = df[column + item].replace([None], '0')\n\n    \n    df['junction_aa'] = df['junction_aa'].apply(encode_atchley_kmer)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:15:24.607685Z","iopub.execute_input":"2025-11-09T22:15:24.608300Z","iopub.status.idle":"2025-11-09T22:15:24.615681Z","shell.execute_reply.started":"2025-11-09T22:15:24.608269Z","shell.execute_reply":"2025-11-09T22:15:24.614583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:43.283030Z","iopub.execute_input":"2025-11-09T21:28:43.283858Z","iopub.status.idle":"2025-11-09T21:28:43.314104Z","shell.execute_reply.started":"2025-11-09T21:28:43.283828Z","shell.execute_reply":"2025-11-09T21:28:43.313308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\n\nresult_list = []\n\nX_train = pd.merge(X_train, y_train, left_index=True, right_index=True)\n\nfor index, row in tqdm(X_train.iterrows(), total=len(X_train), desc=\"Processing files\"):\n    filebase = \"/kaggle/input/adaptive-immune-profiling-challenge-2025/train_datasets/train_datasets/train_dataset_8/\"\n    sdf = pd.read_csv(filebase + row['filename'], sep='\\t')\n    # sample data from sequencing - datasets are too large for TabPFN\n    sdf = process_sequencing_file(sdf.sample(10))\n    sdf = sdf.assign(\n        original_index=index,\n        **row.to_dict()\n    )\n    result_list.append(sdf)\n\nresult = pd.concat(result_list, ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:28:43.314916Z","iopub.execute_input":"2025-11-09T21:28:43.315232Z","iopub.status.idle":"2025-11-09T21:30:44.425300Z","shell.execute_reply.started":"2025-11-09T21:28:43.315210Z","shell.execute_reply":"2025-11-09T21:30:44.424427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:30:44.426091Z","iopub.execute_input":"2025-11-09T21:30:44.426320Z","iopub.status.idle":"2025-11-09T21:30:44.448779Z","shell.execute_reply.started":"2025-11-09T21:30:44.426301Z","shell.execute_reply":"2025-11-09T21:30:44.448069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:30:44.449877Z","iopub.execute_input":"2025-11-09T21:30:44.450245Z","iopub.status.idle":"2025-11-09T21:30:44.512390Z","shell.execute_reply.started":"2025-11-09T21:30:44.450207Z","shell.execute_reply":"2025-11-09T21:30:44.511575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:30:44.513169Z","iopub.execute_input":"2025-11-09T21:30:44.513396Z","iopub.status.idle":"2025-11-09T21:30:44.520167Z","shell.execute_reply.started":"2025-11-09T21:30:44.513377Z","shell.execute_reply":"2025-11-09T21:30:44.519357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tabpfn_client import init, TabPFNClassifier","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:31:47.445627Z","iopub.execute_input":"2025-11-09T21:31:47.446057Z","iopub.status.idle":"2025-11-09T21:31:48.071567Z","shell.execute_reply.started":"2025-11-09T21:31:47.446019Z","shell.execute_reply":"2025-11-09T21:31:48.070275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result.to_csv('cleaned.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:04:37.146435Z","iopub.execute_input":"2025-11-09T22:04:37.146835Z","iopub.status.idle":"2025-11-09T22:04:37.556462Z","shell.execute_reply.started":"2025-11-09T22:04:37.146807Z","shell.execute_reply":"2025-11-09T22:04:37.555786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nresult = pd.read_csv('/kaggle/input/cleaned-airrml/cleaned.csv')\nresult.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:34:34.939987Z","iopub.execute_input":"2025-11-09T22:34:34.940394Z","iopub.status.idle":"2025-11-09T22:34:35.102853Z","shell.execute_reply.started":"2025-11-09T22:34:34.940366Z","shell.execute_reply":"2025-11-09T22:34:35.101898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result_filtered = result[['junction_aa', 'templates', 'v_callf', 'v_calls', 'j_callf', 'j_calls', 'label_positive', 'age', 'sex_F', 'sex_M']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:34:37.329393Z","iopub.execute_input":"2025-11-09T22:34:37.329793Z","iopub.status.idle":"2025-11-09T22:34:37.337005Z","shell.execute_reply.started":"2025-11-09T22:34:37.329766Z","shell.execute_reply":"2025-11-09T22:34:37.335950Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result_filtered.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:34:38.819756Z","iopub.execute_input":"2025-11-09T22:34:38.820458Z","iopub.status.idle":"2025-11-09T22:34:38.831943Z","shell.execute_reply.started":"2025-11-09T22:34:38.820429Z","shell.execute_reply":"2025-11-09T22:34:38.830869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X, y = result_filtered.drop(columns=['label_positive']), result_filtered['label_positive']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:34:41.051251Z","iopub.execute_input":"2025-11-09T22:34:41.052277Z","iopub.status.idle":"2025-11-09T22:34:41.058810Z","shell.execute_reply.started":"2025-11-09T22:34:41.052241Z","shell.execute_reply":"2025-11-09T22:34:41.057530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -U tabpfn_client","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T21:37:50.195381Z","iopub.execute_input":"2025-11-09T21:37:50.195738Z","iopub.status.idle":"2025-11-09T21:38:01.319450Z","shell.execute_reply.started":"2025-11-09T21:37:50.195712Z","shell.execute_reply":"2025-11-09T21:38:01.318409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tabpfn_client import init, TabPFNClassifier","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:34:46.635537Z","iopub.execute_input":"2025-11-09T22:34:46.636300Z","iopub.status.idle":"2025-11-09T22:34:46.640869Z","shell.execute_reply.started":"2025-11-09T22:34:46.636272Z","shell.execute_reply":"2025-11-09T22:34:46.639706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X[['junction_aa_1', 'junction_aa_2', 'junction_aa_3', 'junction_aa_4', 'junction_aa_5']] = X['junction_aa'].apply(\n    lambda x: pd.Series([float(i) for i in x.strip(\"[]\").split()])\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:37:57.480432Z","iopub.execute_input":"2025-11-09T22:37:57.481467Z","iopub.status.idle":"2025-11-09T22:37:58.446202Z","shell.execute_reply.started":"2025-11-09T22:37:57.481430Z","shell.execute_reply":"2025-11-09T22:37:58.445457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:38:00.030695Z","iopub.execute_input":"2025-11-09T22:38:00.031035Z","iopub.status.idle":"2025-11-09T22:38:00.046457Z","shell.execute_reply.started":"2025-11-09T22:38:00.031007Z","shell.execute_reply":"2025-11-09T22:38:00.045246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = X.drop(columns=['junction_aa'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:38:32.913210Z","iopub.execute_input":"2025-11-09T22:38:32.914106Z","iopub.status.idle":"2025-11-09T22:38:32.919705Z","shell.execute_reply.started":"2025-11-09T22:38:32.914077Z","shell.execute_reply":"2025-11-09T22:38:32.918930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = TabPFNClassifier()\nmodel.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:38:35.044418Z","iopub.execute_input":"2025-11-09T22:38:35.045144Z","iopub.status.idle":"2025-11-09T22:38:36.661824Z","shell.execute_reply.started":"2025-11-09T22:38:35.045117Z","shell.execute_reply":"2025-11-09T22:38:36.660758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\nmeta_dataset7 = pd.read_csv(\"/kaggle/input/adaptive-immune-profiling-challenge-2025/train_datasets/train_datasets/train_dataset_7/metadata.csv\")\nmeta_dataset7.head()\nsex_encoding = pd.get_dummies(meta_dataset7['sex'], prefix=\"sex\")\nmeta_dataset7 = pd.merge(meta_dataset7, sex_encoding, left_index=True, right_index=True)\nmeta_dataset7 = meta_dataset7.drop(columns=['sex'])\n\nresult_list = []\n\nfor index, row in tqdm(meta_dataset7.iterrows(), total=len(meta_dataset7), desc=\"Processing files\"):\n    filebase = \"/kaggle/input/adaptive-immune-profiling-challenge-2025/train_datasets/train_datasets/train_dataset_7/\"\n    sdf = pd.read_csv(filebase + row['filename'], sep='\\t')\n    # sample data from sequencing - datasets are too large for TabPFN\n    sdf = process_sequencing_file(sdf.sample(10), targets=['v', 'j'])\n    sdf = sdf.assign(\n        original_index=index,\n        **row.to_dict()\n    )\n    result_list.append(sdf)\n\nresult_7 = pd.concat(result_list, ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:15:55.544790Z","iopub.execute_input":"2025-11-09T22:15:55.545578Z","iopub.status.idle":"2025-11-09T22:18:08.043934Z","shell.execute_reply.started":"2025-11-09T22:15:55.545554Z","shell.execute_reply":"2025-11-09T22:18:08.042851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result_7 = result_7.drop(columns=['repertoire_id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:23:19.510994Z","iopub.execute_input":"2025-11-09T22:23:19.511335Z","iopub.status.idle":"2025-11-09T22:23:19.517818Z","shell.execute_reply.started":"2025-11-09T22:23:19.511310Z","shell.execute_reply":"2025-11-09T22:23:19.516925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result_7 = result_7.drop(columns=['sequencing_run_id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:23:44.214517Z","iopub.execute_input":"2025-11-09T22:23:44.215381Z","iopub.status.idle":"2025-11-09T22:23:44.221296Z","shell.execute_reply.started":"2025-11-09T22:23:44.215353Z","shell.execute_reply":"2025-11-09T22:23:44.220411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result_7.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:23:46.770370Z","iopub.execute_input":"2025-11-09T22:23:46.771200Z","iopub.status.idle":"2025-11-09T22:23:46.783956Z","shell.execute_reply.started":"2025-11-09T22:23:46.771174Z","shell.execute_reply":"2025-11-09T22:23:46.783055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test, y_test = result_7.drop(columns=['label_positive']), result_7['label_positive']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:24:48.824265Z","iopub.execute_input":"2025-11-09T22:24:48.824578Z","iopub.status.idle":"2025-11-09T22:24:48.830772Z","shell.execute_reply.started":"2025-11-09T22:24:48.824555Z","shell.execute_reply":"2025-11-09T22:24:48.829867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test[['junction_aa_1', 'junction_aa_2', 'junction_aa_3', 'junction_aa_4', 'junction_aa_5']] = X_test['junction_aa'].apply(\n    lambda x: pd.Series([float(i) for i in str(x).strip(\"[]\").split()])\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:39:53.995943Z","iopub.execute_input":"2025-11-09T22:39:53.996300Z","iopub.status.idle":"2025-11-09T22:39:54.724064Z","shell.execute_reply.started":"2025-11-09T22:39:53.996274Z","shell.execute_reply":"2025-11-09T22:39:54.723299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = X_test.drop(columns=['junction_aa'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:40:37.321902Z","iopub.execute_input":"2025-11-09T22:40:37.322251Z","iopub.status.idle":"2025-11-09T22:40:37.329275Z","shell.execute_reply.started":"2025-11-09T22:40:37.322227Z","shell.execute_reply":"2025-11-09T22:40:37.328064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = X_test.rename(columns={\"sex_female\": \"sex_F\", \"sex_male\": \"sex_M\"})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:42:08.233054Z","iopub.execute_input":"2025-11-09T22:42:08.233807Z","iopub.status.idle":"2025-11-09T22:42:08.240175Z","shell.execute_reply.started":"2025-11-09T22:42:08.233774Z","shell.execute_reply":"2025-11-09T22:42:08.239244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:42:23.142855Z","iopub.execute_input":"2025-11-09T22:42:23.143164Z","iopub.status.idle":"2025-11-09T22:42:28.492269Z","shell.execute_reply.started":"2025-11-09T22:42:23.143141Z","shell.execute_reply":"2025-11-09T22:42:28.491396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"probabilities = model.predict_proba(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:42:34.466134Z","iopub.execute_input":"2025-11-09T22:42:34.466828Z","iopub.status.idle":"2025-11-09T22:42:39.308717Z","shell.execute_reply.started":"2025-11-09T22:42:34.466802Z","shell.execute_reply":"2025-11-09T22:42:39.307697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score, confusion_matrix\nprint(confusion_matrix(y_test, predictions)) \nprint(roc_auc_score(y_test, predictions))\n# If asked to predict with full data, then model is 1.0 ROC AUC \n# This is very much expected since we test on data model has already seen - what if we test on another dataset? \n# If asked to predict with limited data, model is 0.5 ROC AUC\n# If data is limited to set 7 + a lot of data model had not seen, then the ROC AUC is 0.485 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T22:43:06.814480Z","iopub.execute_input":"2025-11-09T22:43:06.815275Z","iopub.status.idle":"2025-11-09T22:43:06.827098Z","shell.execute_reply.started":"2025-11-09T22:43:06.815248Z","shell.execute_reply":"2025-11-09T22:43:06.826077Z"}},"outputs":[],"execution_count":null}]}