{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":106680,"databundleVersionId":13374319,"sourceType":"competition"},{"sourceId":14182189,"sourceType":"datasetVersion","datasetId":9041351}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Path to the competition data in Kaggle notebooks\ndata_path = '/kaggle/input/adaptive-immune-profiling-challenge-2025'  # Adjust if the folder name is slightly different\n\nprint(\"Folders and files:\")\nfor root, dirs, files in os.walk(data_path):\n    level = root.replace(data_path, '').count(os.sep)\n    indent = ' ' * 4 * level\n    print(f\"{indent}{os.path.basename(root)}/\")\n    for f in files[:10]:  # Show first 10 files\n        print(f\"{indent}    {f}\")\n    if len(files) > 10:\n        print(f\"{indent}    ... and {len(files)-10} more files\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T17:13:19.709435Z","iopub.execute_input":"2025-12-16T17:13:19.710221Z","iopub.status.idle":"2025-12-16T17:13:21.985752Z","shell.execute_reply.started":"2025-12-16T17:13:19.710186Z","shell.execute_reply":"2025-12-16T17:13:21.984597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# List all available input directories\ninput_path = '/kaggle/input'\nprint(\"All competition data folders in /kaggle/input/:\")\nfor folder in os.listdir(input_path):\n    full_path = os.path.join(input_path, folder)\n    if os.path.isdir(full_path):\n        print(f\"- {folder}\")\n        # Show subcontents\n        sub = os.listdir(full_path)\n        print(f\"  Subfolders/files (first 20): {sub[:20]}\")\n        if len(sub) > 20:\n            print(\"  ... more\")\n\n# Once you see the correct folder (look for one with 'train_datasets' or 'airr' in name/sub), copy it here:\n# data_path = '/kaggle/input/YOUR_EXACT_FOLDER_NAME_HERE'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T17:20:34.429901Z","iopub.execute_input":"2025-12-16T17:20:34.430293Z","iopub.status.idle":"2025-12-16T17:20:34.442362Z","shell.execute_reply.started":"2025-12-16T17:20:34.430267Z","shell.execute_reply":"2025-12-16T17:20:34.440850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndata_path = '/kaggle/input/adaptive-immune-profiling-challenge-2025'\n\nnested_train = os.path.join(data_path, 'train_datasets', 'train_datasets')\nnested_test = os.path.join(data_path, 'test_datasets', 'test_datasets')\n\nprint(\"Real train datasets folders:\")\nprint(os.listdir(nested_train))\n\nprint(\"\\nExample inside train_dataset_1:\")\nexample = os.path.join(nested_train, 'train_dataset_1')\nprint(os.listdir(example)[:20])  # Should show metadata.csv and .tsv files","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T17:31:47.331516Z","iopub.execute_input":"2025-12-16T17:31:47.332190Z","iopub.status.idle":"2025-12-16T17:31:47.341170Z","shell.execute_reply.started":"2025-12-16T17:31:47.332164Z","shell.execute_reply":"2025-12-16T17:31:47.340270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\ndata_path = '/kaggle/input/adaptive-immune-profiling-challenge-2025'\n\n# Adjust these based on Step 1 output\ntrain_base = os.path.join(data_path, 'train_datasets', 'train_datasets')  # Likely nested\ntest_base = os.path.join(data_path, 'test_datasets', 'test_datasets')     # Same for test\n\n# List real datasets\nprint(\"Real train datasets:\", os.listdir(train_base))\n\ndef load_dataset(dataset_name, is_train=True):\n    base = train_base if is_train else test_base\n    folder = os.path.join(base, dataset_name)\n    \n    print(f\"Loading {dataset_name} from {folder}\")\n    \n    # Metadata inside the dataset folder\n    meta_path = os.path.join(folder, 'metadata.csv')\n    metadata = pd.read_csv(meta_path) if os.path.exists(meta_path) else None\n    if metadata is not None:\n        print(f\"Metadata: {len(metadata)} rows, columns: {metadata.columns.tolist()}\")\n    \n    # Load TSVs\n    repertoires = []\n    for file in os.listdir(folder):\n        if file.endswith('.tsv'):\n            rep_id = os.path.splitext(file)[0]\n            df = pd.read_csv(os.path.join(folder, file), sep='\\t')\n            df['repertoire_id'] = rep_id\n            repertoires.append(df)\n            if len(repertoires) > 5:  # Test mode - remove for full load\n                break\n    \n    df_full = pd.concat(repertoires, ignore_index=True)\n    \n    if metadata is not None:\n        # filename column likely has .tsv, so strip\n        metadata['repertoire_id'] = metadata['filename'].str.replace('.tsv', '', regex=False)\n        df_full = df_full.merge(metadata[['repertoire_id', 'label_positive']], on='repertoire_id', how='left')\n    \n    return df_full\n\n# Test\ntrain1 = load_dataset('train_dataset_1', is_train=True)\nprint(train1.head())\nprint(\"Columns:\", train1.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T17:31:52.775007Z","iopub.execute_input":"2025-12-16T17:31:52.775301Z","iopub.status.idle":"2025-12-16T17:31:52.944748Z","shell.execute_reply.started":"2025-12-16T17:31:52.775280Z","shell.execute_reply":"2025-12-16T17:31:52.944019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.preprocessing import normalize\nfrom sklearn.linear_model import LogisticRegression\nimport numpy as np\nfrom scipy.sparse import vstack  # If needed later","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T17:35:41.712996Z","iopub.execute_input":"2025-12-16T17:35:41.713856Z","iopub.status.idle":"2025-12-16T17:35:42.333064Z","shell.execute_reply.started":"2025-12-16T17:35:41.713831Z","shell.execute_reply":"2025-12-16T17:35:42.332477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.preprocessing import normalize\nfrom sklearn.linear_model import LogisticRegression\nimport numpy as np\nimport os\nimport pandas as pd\n\n# Your existing data_path, train_base, test_base, load_dataset function here...\n\nall_test_preds = []\nall_explanations = []\n\ntrain_datasets = [f'train_dataset_{i}' for i in range(1, 9)]\n\n# List all test datasets once\ntest_datasets = os.listdir(test_base)\n\nfor train_ds in train_datasets:\n    print(f\"\\nProcessing {train_ds}...\")\n    train_df = load_dataset(train_ds, is_train=True)\n    \n    # Determine count column (templates is common in Adaptive data)\n    count_col = 'templates' if 'templates' in train_df.columns else 'duplicate_count'\n    if count_col not in train_df.columns:\n        count_col = None  # No counts, use uniform\n    \n    if count_col:\n        train_df['freq'] = train_df[count_col] / train_df.groupby('repertoire_id')[count_col].transform('sum')\n    else:\n        train_df['freq'] = 1.0 / train_df.groupby('repertoire_id').transform('size')\n    \n    # Grouped texts for vectorizer (fixed deprecation)\n    grouped = train_df.groupby('repertoire_id', group_keys=False).apply(\n        lambda g: ' '.join(g['junction_aa'].astype(str)), include_groups=False\n    )\n    \n    y_train = train_df.groupby('repertoire_id')['label_positive'].first().reindex(grouped.index).values\n    \n    class_info = np.unique(y_train, return_counts=True)\n    print(f\"{train_ds} classes: {class_info}\")\n    \n    if len(class_info[0]) < 2:\n        print(f\"Single class detected ({class_info[0][0]}). Skipping model training.\")\n        \n        # Fallback explanations: top 50k most frequent unique sequences\n        if count_col:\n            top_seqs = (train_df.groupby(['junction_aa', 'v_call', 'j_call'])[count_col]\n                        .sum()\n                        .sort_values(ascending=False)\n                        .head(50000)\n                        .reset_index())\n        else:\n            top_seqs = (train_df[['junction_aa', 'v_call', 'j_call']]\n                        .value_counts()\n                        .head(50000)\n                        .reset_index()\n                        .drop(columns='count'))\n        \n        top_seqs['IDdataset'] = train_ds\n        all_explanations.append(top_seqs[['IDdataset', 'junction_aa', 'v_call', 'j_call']])\n        \n        # Default predictions for linked tests: probability = majority class\n        default_prob = 1.0 if class_info[0][0] else 0.0\n        \n        # Find linked test datasets (adjust naming pattern if needed)\n        linked_tests = [td for td in test_datasets if td.startswith(train_ds.replace('train_dataset_', 'test_dataset_'))]\n        for test_ds in linked_tests:\n            test_df = load_dataset(test_ds, is_train=False)\n            test_rep_ids = test_df['repertoire_id'].unique()\n            for rep_id in test_rep_ids:\n                all_test_preds.append({'repertoire_id': rep_id, 'label_positive_probability': default_prob})\n        \n        continue  # Skip to next train dataset\n    \n    # Normal case: train model\n    vectorizer = CountVectorizer(analyzer='char', ngram_range=(3,3), lowercase=False)\n    X_train = vectorizer.fit_transform(grouped.values)\n    X_train = normalize(X_train, norm='l1')\n    \n    model = LogisticRegression(max_iter=1000, class_weight='balanced')\n    model.fit(X_train, y_train)\n    \n    # Predictions on linked tests\n    linked_tests = [td for td in test_datasets if td.startswith(train_ds.replace('train_dataset_', 'test_dataset_'))]\n    for test_ds in linked_tests:\n        test_df = load_dataset(test_ds, is_train=False)\n        grouped_test = test_df.groupby('repertoire_id', group_keys=False).apply(\n            lambda g: ' '.join(g['junction_aa'].astype(str)), include_groups=False\n        )\n        X_test = vectorizer.transform(grouped_test.values)\n        X_test = normalize(X_test, norm='l1')\n        probs = model.predict_proba(X_test)[:, 1]\n        \n        for rep_id, prob in zip(grouped_test.index, probs):\n            all_test_preds.append({'repertoire_id': rep_id, 'label_positive_probability': prob})\n    \n    # Explanations: score unique sequences by k-mer coefficients\n    coef = model.coef_[0]\n    unique_seqs = train_df[['junction_aa', 'v_call', 'j_call']].drop_duplicates()\n    \n    def seq_score(seq):\n        if len(seq) < 3:\n            return 0.0\n        kmers = [' '.join([seq[i:i+3] for i in range(len(seq)-2)])]\n        counts = vectorizer.transform(kmers)\n        return np.dot(counts.toarray()[0], coef)\n    \n    unique_seqs['importance'] = unique_seqs['junction_aa'].apply(seq_score)\n    unique_seqs = unique_seqs.sort_values('importance', ascending=False).head(50000)\n    unique_seqs['IDdataset'] = train_ds\n    all_explanations.append(unique_seqs[['IDdataset', 'junction_aa', 'v_call', 'j_call']])\n\n# Final submission assembly (after loop)\npred_df = pd.DataFrame(all_test_preds)\npred_df['junction_aa'] = -999.0\npred_df['v_call'] = -999.0\npred_df['j_call'] = -999.0\n\nexpl_df = pd.concat(all_explanations, ignore_index=True)\nexpl_df['label_positive_probability'] = -999.0\nexpl_df['repertoire_id'] = ''  # Or leave blank if not needed\n\nsubmission = pd.concat([pred_df, expl_df], ignore_index=True)\nsubmission = submission[['repertoire_id', 'label_positive_probability', 'junction_aa', 'v_call', 'j_call']]\n\n# Fill any missing with -999 if needed\nsubmission.fillna(-999.0, inplace=True)\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission created! Rows:\", len(submission))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T17:38:41.349609Z","iopub.execute_input":"2025-12-16T17:38:41.350183Z","iopub.status.idle":"2025-12-16T17:41:49.821972Z","shell.execute_reply.started":"2025-12-16T17:38:41.350161Z","shell.execute_reply":"2025-12-16T17:41:49.821132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# After the full loop over train_datasets\n\nprint(f\"Collected {len(all_test_preds)} predictions\")\nprint(f\"Collected explanations from {len(all_explanations)} datasets\")\n\n# Predictions dataframe (4213 rows expected)\npred_df = pd.DataFrame(all_test_preds)\npred_df = pred_df[['repertoire_id', 'label_positive_probability']]\npred_df['junction_aa'] = -999.0\npred_df['v_call'] = -999.0\npred_df['j_call'] = -999.0\n\n# Explanations (400,000 rows: 50k per 8 datasets)\nexpl_df = pd.concat(all_explanations, ignore_index=True)\nexpl_df['label_positive_probability'] = -999.0\nexpl_df['repertoire_id'] = -999.0  # Or leave as '' if not required\n\n# Full submission\nsubmission = pd.concat([pred_df, expl_df], ignore_index=True)\nsubmission = submission[['repertoire_id', 'label_positive_probability', 'junction_aa', 'v_call', 'j_call']]\n\n# Ensure exact 404213 rows and no NaNs\nsubmission.fillna(-999.0, inplace=True)\nprint(\"Final submission rows:\", len(submission))\nsubmission.head(10)\nsubmission.tail(10)\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\"submission.csv created – download and submit on Kaggle!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T17:43:18.004470Z","iopub.execute_input":"2025-12-16T17:43:18.004979Z","iopub.status.idle":"2025-12-16T17:43:19.326082Z","shell.execute_reply.started":"2025-12-16T17:43:18.004956Z","shell.execute_reply":"2025-12-16T17:43:19.324971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.listdir('/kaggle/working'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:29:13.354971Z","iopub.execute_input":"2025-12-16T18:29:13.355693Z","iopub.status.idle":"2025-12-16T18:29:13.360457Z","shell.execute_reply.started":"2025-12-16T18:29:13.355671Z","shell.execute_reply":"2025-12-16T18:29:13.359554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Load the sample submission file (template with exact format)\nsample_path = '/kaggle/input/adaptive-immune-profiling-challenge-2025/sample_submissions.csv'\nsample = pd.read_csv(sample_path)\n\nprint(\"Sample columns:\", sample.columns.tolist())\nprint(\"Sample rows:\", len(sample))  # Should be 404213\n\n# Fill everything with -999.0 (float)\nsample[:] = -999.0\n\n# Number of prediction rows (from competition description)\nnum_pred_rows = 4213\n\n# Your predictions\npred_df = pd.DataFrame(all_test_preds)\nprint(f\"You have {len(pred_df)} predictions\")\n\n# Fill prediction probabilities (first 4213 rows)\nprobs = pred_df['label_positive_probability'].reindex(range(num_pred_rows)).fillna(0.5).values\nsample.iloc[:num_pred_rows, sample.columns.get_loc('label_positive_probability')] = probs\n\nprint(f\"Filled {num_pred_rows} prediction probabilities\")\n\n# Start filling explanations after predictions\ncurrent_row = num_pred_rows\n\n# Concat all explanations\nexpl_df = pd.concat(all_explanations, ignore_index=True)\n\n# Fill 50,000 rows for each of the 8 training datasets\nfor i in range(1, 9):\n    dataset_name = f'train_dataset_{i}'\n    print(f\"Processing explanations for {dataset_name}...\")\n    \n    # Select this dataset's explanations\n    if 'IDdataset' in expl_df.columns:\n        dataset_expl = expl_df[expl_df['IDdataset'] == dataset_name][['junction_aa', 'v_call', 'j_call']]\n    else:\n        # If no IDdataset, take next 50k chunk\n        start_idx = (i-1) * 50000\n        dataset_expl = expl_df.iloc[start_idx:start_idx+50000][['junction_aa', 'v_call', 'j_call']]\n    \n    dataset_expl = dataset_expl.copy()\n    \n    # Pad with -999.0 if less than 50,000\n    if len(dataset_expl) < 50000:\n        pad_rows = 50000 - len(dataset_expl)\n        pad_df = pd.DataFrame({\n            'junction_aa': [-999.0] * pad_rows,\n            'v_call': [-999.0] * pad_rows,\n            'j_call': [-999.0] * pad_rows\n        })\n        dataset_expl = pd.concat([dataset_expl, pad_df], ignore_index=True)\n    \n    # Take only top 50,000\n    dataset_expl = dataset_expl.head(50000)\n    \n    # Fill into sample\n    end_row = current_row + 50000\n    sample.iloc[current_row:end_row, sample.columns.get_loc('junction_aa')] = dataset_expl['junction_aa'].values\n    sample.iloc[current_row:end_row, sample.columns.get_loc('v_call')] = dataset_expl['v_call'].values\n    sample.iloc[current_row:end_row, sample.columns.get_loc('j_call')] = dataset_expl['j_call'].values\n    \n    current_row += 50000\n\n# Save final submission\nsample.to_csv('submission.csv', index=False)\n\nprint(\"SUCCESS! submission.csv is ready with exact 404213 rows\")\nprint(\"Download and submit it now!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:34:54.802761Z","iopub.execute_input":"2025-12-16T18:34:54.803330Z","iopub.status.idle":"2025-12-16T18:34:57.209095Z","shell.execute_reply.started":"2025-12-16T18:34:54.803307Z","shell.execute_reply":"2025-12-16T18:34:57.208189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load the sample (template with exact format)\nsample = pd.read_csv('/kaggle/input/adaptive-immune-profiling-challenge-2025/sample_submissions.csv')\n\nprint(\"Columns:\", sample.columns.tolist())\nprint(\"Rows:\", len(sample))  # Must be 404213\n\n# Fill with -999.0\nsample[:] = -999.0\n\n# Fill your predictions (first rows – probability column)\nnum_preds = len(all_test_preds)\nif num_preds > 0:\n    probs = pd.DataFrame(all_test_preds)['label_positive_probability'].values\n    sample.iloc[:num_preds, sample.columns.get_loc('label_positive_probability')] = probs\n\n# Fill explanations if possible (simple way – spread your all_explanations across the explanation rows)\nexpl_start = 4213\nexpl_rows = len(sample) - expl_start\nexpl_df = pd.concat(all_explanations, ignore_index=True)\n\nif len(expl_df) > 0:\n    # Repeat or pad to fill explanation rows\n    expl_filled = pd.concat([expl_df] * (expl_rows // len(expl_df) + 1), ignore_index=True).head(expl_rows)\n    sample.iloc[expl_start:, sample.columns.get_loc('junction_aa')] = expl_filled['junction_aa'].values\n    sample.iloc[expl_start:, sample.columns.get_loc('v_call')] = expl_filled['v_call'].values\n    sample.iloc[expl_start:, sample.columns.get_loc('j_call')] = expl_filled['j_call'].values\n\n# Save\nsample.to_csv('submission.csv', index=False)\n\nprint(\"New submission.csv ready! Rows:\", len(sample))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:44:52.749603Z","iopub.execute_input":"2025-12-16T18:44:52.749882Z","iopub.status.idle":"2025-12-16T18:44:54.946126Z","shell.execute_reply.started":"2025-12-16T18:44:52.749860Z","shell.execute_reply":"2025-12-16T18:44:54.945323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load sample (template with unique IDs)\nsample = pd.read_csv('/kaggle/input/adaptive-immune-profiling-challenge-2025/sample_submissions.csv')\n\nprint(\"Columns:\", sample.columns.tolist())  # Likely ['ID', 'label_positive_probability', 'junction_aa', 'v_call', 'j_call'] or similar\n\n# Overwrite everything with -999.0 (safe)\nsample.iloc[:, 1:] = -999.0  # Keep first column (ID) intact, fill others\n\n# Fill predictions (first 4213 rows)\nnum_preds = 4213\nprobs = pd.DataFrame(all_test_preds)['label_positive_probability'].values\nif len(probs) < num_preds:\n    probs = list(probs) + [0.5] * (num_preds - len(probs))  # Default\n\nsample.iloc[:num_preds, sample.columns.get_loc('label_positive_probability')] = probs[:num_preds]\n\n# Fill explanations (after predictions, no IDdataset added)\nexpl_start = num_preds\ncurrent_row = expl_start\n\nexpl_df = pd.concat(all_explanations, ignore_index=True)\nexpl_df = expl_df[['junction_aa', 'v_call', 'j_call']]  # Drop any IDdataset column!\n\nfor i in range(8):  # 8 datasets\n    start = i * 50000\n    end = start + 50000\n    dataset_expl = expl_df.iloc[start:end] if len(expl_df) > start else pd.DataFrame()\n    \n    if len(dataset_expl) < 50000:\n        pad = pd.DataFrame({\n            'junction_aa': [-999.0] * (50000 - len(dataset_expl)),\n            'v_call': [-999.0] * (50000 - len(dataset_expl)),\n            'j_call': [-999.0] * (50000 - len(dataset_expl))\n        })\n        dataset_expl = pd.concat([dataset_expl, pad], ignore_index=True)\n    \n    end_row = current_row + 50000\n    sample.iloc[current_row:end_row, sample.columns.get_loc('junction_aa')] = dataset_expl['junction_aa'].values\n    sample.iloc[current_row:end_row, sample.columns.get_loc('v_call')] = dataset_expl['v_call'].values\n    sample.iloc[current_row:end_row, sample.columns.get_loc('j_call')] = dataset_expl['j_call'].values\n    \n    current_row += 50000\n\n# Save\nsample.to_csv('submission.csv', index=False)\n\nprint(\"Fixed – no duplicates! Ready to submit.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:52:58.359288Z","iopub.execute_input":"2025-12-16T18:52:58.359580Z","iopub.status.idle":"2025-12-16T18:53:00.738064Z","shell.execute_reply.started":"2025-12-16T18:52:58.359559Z","shell.execute_reply":"2025-12-16T18:53:00.737356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load sample (has unique ID and correct dataset values)\nsample = pd.read_csv('/kaggle/input/adaptive-immune-profiling-challenge-2025/sample_submissions.csv')\n\nprint(\"Columns:\", sample.columns.tolist())  # Confirm ['ID', 'dataset', ...]\n\n# Fill all fillable columns with -999.0 (keep ID and dataset intact)\nfillable_cols = ['label_positive_probability', 'junction_aa', 'v_call', 'j_call']\nsample[fillable_cols] = -999.0\n\n# Fill predictions (first 4213 rows: label_positive_probability)\nnum_preds = 4213\nprobs = pd.DataFrame(all_test_preds)['label_positive_probability'].values.tolist()\nif len(probs) < num_preds:\n    probs += [0.5] * (num_preds - len(probs))  # Default if missing\n\nsample.iloc[:num_preds, sample.columns.get_loc('label_positive_probability')] = probs[:num_preds]\n\n# Fill explanations (from row 4213 onward)\nexpl_start = num_preds\ncurrent_row = expl_start\n\nexpl_df = pd.concat(all_explanations, ignore_index=True)\n\n# Ensure only the 3 columns, no extra\nexpl_df = expl_df[['junction_aa', 'v_call', 'j_call']]\n\nfor i in range(8):\n    start_idx = i * 50000\n    dataset_expl = expl_df.iloc[start_idx:start_idx+50000]\n    \n    if len(dataset_expl) < 50000:\n        pad = pd.DataFrame({\n            'junction_aa': [-999.0] * (50000 - len(dataset_expl)),\n            'v_call': [-999.0] * (50000 - len(dataset_expl)),\n            'j_call': [-999.0] * (50000 - len(dataset_expl))\n        })\n        dataset_expl = pd.concat([dataset_expl, pad], ignore_index=True)\n    \n    end_row = current_row + 50000\n    sample.iloc[current_row:end_row, sample.columns.get_loc('junction_aa')] = dataset_expl['junction_aa'].values\n    sample.iloc[current_row:end_row, sample.columns.get_loc('v_call')] = dataset_expl['v_call'].values\n    sample.iloc[current_row:end_row, sample.columns.get_loc('j_call')] = dataset_expl['j_call'].values\n    \n    current_row += 50000\n\n# Save\nsample.to_csv('submission.csv', index=False)\n\nprint(\"FINAL fixed submission ready – no duplicates, correct format!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:53:44.854523Z","iopub.execute_input":"2025-12-16T18:53:44.854772Z","iopub.status.idle":"2025-12-16T18:53:47.251453Z","shell.execute_reply.started":"2025-12-16T18:53:44.854755Z","shell.execute_reply":"2025-12-16T18:53:47.250563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink('submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-16T18:54:25.738146Z","iopub.execute_input":"2025-12-16T18:54:25.738869Z","iopub.status.idle":"2025-12-16T18:54:25.744107Z","shell.execute_reply.started":"2025-12-16T18:54:25.738843Z","shell.execute_reply":"2025-12-16T18:54:25.743332Z"}},"outputs":[],"execution_count":null}]}