{"cells": [{"cell_type": "markdown", "metadata": {}, "source": "# Perseptron v2 - 07 Optimized Kaggle Submission Generation\n\nThis notebook creates full-customer Kaggle submission files for the three recommendation models: `tabular_only`, `image_history`, and `late_fusion`.\n\nThe notebook uses a candidate-generation plus reranking pipeline. All customers from `sample_submission.csv` are preserved, while each model scores a compact candidate set instead of the whole article catalog.\n\nOutputs:\n\n- `submission_tabular_only.csv`\n- `submission_image_history.csv`\n- `submission_late_fusion.csv`\n- `submission_generation_summary.md`\n\nThe image-only CNN is not used here because it is a classification/explainability baseline, not a customer-level recommender.", "id": "cell-00"}, {"cell_type": "markdown", "metadata": {}, "source": "## Environment and settings\n\nThis cell discovers Kaggle inputs, sets full-run submission parameters, and keeps all text ASCII to avoid notebook rendering issues.", "id": "cell-01"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "from pathlib import Path\nimport gc\nimport json\nimport os\nimport time\nimport warnings\n\nimport numpy as np\nimport pandas as pd\nimport torch\nfrom torch import nn\nfrom tqdm.auto import tqdm\n\nwarnings.filterwarnings('ignore')\n\nIS_KAGGLE = Path('/kaggle').exists()\nWORK_DIR = Path('/kaggle/working') if IS_KAGGLE else Path.cwd()\nINPUT_ROOTS = [Path('/kaggle/input')] if IS_KAGGLE else [Path.cwd(), Path.cwd() / 'artifacts' / 'final', Path.cwd() / 'data']\n\n# Direct full run. Change to True only for local/debug checks.\nSMOKE_RUN = False\nSMOKE_CUSTOMERS = 1000\n\nTOP_K = 12\nPOPULAR_CANDIDATES = 80\nLAST_7D_CANDIDATES = 80\nLAST_30D_CANDIDATES = 80\nGLOBAL_CANDIDATES = 80\nRECENT_PURCHASE_CANDIDATES = 24\nSIMILAR_PRODUCT_TYPE_CANDIDATES = 16\nSIMILAR_GARMENT_GROUP_CANDIDATES = 16\nCUSTOMER_CANDIDATE_LIMIT = 180\nINFERENCE_CUSTOMER_BATCH_SIZE = 2048\nINFERENCE_SCORE_BATCH_SIZE = 8192\nVISUAL_FEATURE_BATCH_SIZE = 32768\nPROFILE_BUILD_BATCH_SIZE = 50000\n\nMODEL_NAMES = ['tabular_only', 'image_history', 'late_fusion']\nMODEL_BASENAMES = {\n    'tabular_only': ['tabular_only.pt', 'tabular_only_fold0.pt'],\n    'image_history': ['image_history.pt', 'image_history_fold0.pt'],\n    'late_fusion': ['late_fusion.pt', 'late_fusion_fold0.pt'],\n}\n\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nSTART_TIME = time.time()\n\ndef log(message):\n    print(f'[{time.strftime(\"%H:%M:%S\")}] {message}', flush=True)\n\ndef find_file(filename, required=True):\n    candidates = []\n    for root in INPUT_ROOTS:\n        if not root.exists():\n            continue\n        candidates.extend(root.rglob(filename))\n    if not candidates and Path(filename).exists():\n        candidates.append(Path(filename))\n    if not candidates:\n        if required:\n            raise FileNotFoundError(f'{filename} not found. Check Kaggle Add Data inputs.')\n        return None\n    return sorted(candidates, key=lambda path: (len(str(path)), str(path)))[0]\n\ndef find_first_file(filenames, required=True):\n    for filename in filenames:\n        path = find_file(filename, required=False)\n        if path is not None:\n            return path\n    if required:\n        raise FileNotFoundError(f'None of these files were found: {filenames}')\n    return None\n\nTRANSACTIONS_PATH = find_file('transactions_train.csv')\nCUSTOMERS_PATH = find_file('customers.csv')\nARTICLES_PATH = find_file('articles.csv')\nSAMPLE_SUBMISSION_PATH = find_file('sample_submission.csv')\nEMBEDDINGS_PATH = find_file('article_image_embeddings_popular.npy')\nEMBEDDING_IDS_PATH = find_file('article_image_embedding_ids_popular.csv')\nMODEL_PATHS = {name: find_first_file(filenames) for name, filenames in MODEL_BASENAMES.items()}\n\nprint('work_dir:', WORK_DIR)\nprint('device:', DEVICE)\nif torch.cuda.is_available():\n    print('cuda devices:', torch.cuda.device_count(), [torch.cuda.get_device_name(i) for i in range(torch.cuda.device_count())])\nprint('transactions:', TRANSACTIONS_PATH)\nprint('customers:', CUSTOMERS_PATH)\nprint('articles:', ARTICLES_PATH)\nprint('sample_submission:', SAMPLE_SUBMISSION_PATH)\nprint('embeddings:', EMBEDDINGS_PATH)\nprint('embedding ids:', EMBEDDING_IDS_PATH)\nprint('models:', MODEL_PATHS)", "id": "cell-02"}, {"cell_type": "markdown", "metadata": {}, "source": "## Model classes and encoders\n\nThese definitions rebuild the checkpoint architectures and encode tabular metadata without importing project source files.", "id": "cell-03"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "def embedding_dim(size: int) -> int:\n    return min(50, max(4, int(size**0.25 * 8)))\n\nclass TabularOnlyMLP(nn.Module):\n    def __init__(self, numeric_dim: int, category_sizes: list[int]):\n        super().__init__()\n        self.embeddings = nn.ModuleList([nn.Embedding(size, embedding_dim(size)) for size in category_sizes])\n        cat_dim = sum(embedding.embedding_dim for embedding in self.embeddings)\n        self.net = nn.Sequential(\n            nn.Linear(numeric_dim + cat_dim, 256), nn.ReLU(), nn.Dropout(0.2),\n            nn.Linear(256, 128), nn.ReLU(), nn.Dropout(0.1), nn.Linear(128, 1),\n        )\n\n    def forward(self, numeric, categorical):\n        embedded = [emb(categorical[:, idx]) for idx, emb in enumerate(self.embeddings)]\n        return self.net(torch.cat([numeric, *embedded], dim=1)).squeeze(1)\n\nclass ImageHistoryMLP(nn.Module):\n    def __init__(self, image_dim: int):\n        super().__init__()\n        self.net = nn.Sequential(\n            nn.Linear(image_dim * 2 + 2, 512), nn.ReLU(), nn.Dropout(0.25),\n            nn.Linear(512, 128), nn.ReLU(), nn.Dropout(0.1), nn.Linear(128, 1),\n        )\n\n    def forward(self, article_embedding, profile_embedding, visual_similarity, visual_history_count):\n        visual_extra = torch.stack([visual_similarity, visual_history_count], dim=1)\n        return self.net(torch.cat([article_embedding, profile_embedding, visual_extra], dim=1)).squeeze(1)\n\nclass MultimodalLateFusion(nn.Module):\n    def __init__(self, numeric_dim: int, category_sizes: list[int], image_dim: int):\n        super().__init__()\n        self.embeddings = nn.ModuleList([nn.Embedding(size, embedding_dim(size)) for size in category_sizes])\n        cat_dim = sum(embedding.embedding_dim for embedding in self.embeddings)\n        self.tabular_branch = nn.Sequential(\n            nn.Linear(numeric_dim + cat_dim, 256), nn.ReLU(), nn.Dropout(0.2),\n            nn.Linear(256, 128), nn.ReLU(),\n        )\n        self.visual_branch = nn.Sequential(\n            nn.Linear(image_dim * 2 + 2, 512), nn.ReLU(), nn.Dropout(0.25),\n            nn.Linear(512, 128), nn.ReLU(),\n        )\n        self.fusion_head = nn.Sequential(nn.Linear(256, 128), nn.ReLU(), nn.Dropout(0.1), nn.Linear(128, 1))\n\n    def forward(self, numeric, categorical, article_embedding, profile_embedding, visual_similarity, visual_history_count):\n        embedded = [emb(categorical[:, idx]) for idx, emb in enumerate(self.embeddings)]\n        tabular = self.tabular_branch(torch.cat([numeric, *embedded], dim=1))\n        visual_extra = torch.stack([visual_similarity, visual_history_count], dim=1)\n        visual = self.visual_branch(torch.cat([article_embedding, profile_embedding, visual_extra], dim=1))\n        return self.fusion_head(torch.cat([tabular, visual], dim=1)).squeeze(1)\n\ndef build_model(model_name: str, metadata: dict, image_dim: int):\n    category_sizes = [len(metadata['category_maps'][column]) for column in metadata['categorical_features']]\n    numeric_dim = len(metadata['numeric_features'])\n    if model_name == 'tabular_only':\n        return TabularOnlyMLP(numeric_dim, category_sizes)\n    if model_name == 'image_history':\n        return ImageHistoryMLP(image_dim)\n    if model_name == 'late_fusion':\n        return MultimodalLateFusion(numeric_dim, category_sizes, image_dim)\n    raise ValueError(f'Unknown model: {model_name}')\n\ndef load_checkpoint_model(path: Path, device: torch.device):\n    checkpoint = torch.load(path, map_location=device, weights_only=False)\n    model = build_model(checkpoint['model_name'], checkpoint['metadata'], checkpoint['image_dim']).to(device)\n    model.load_state_dict(checkpoint['state_dict'])\n    model.eval()\n    return checkpoint, model\n\ndef encode_tabular(frame: pd.DataFrame, metadata: dict):\n    numeric_columns = []\n    for column in metadata['numeric_features']:\n        values = pd.to_numeric(frame[column], errors='coerce').fillna(metadata['numeric_mean'][column]).astype('float32')\n        values = (values - metadata['numeric_mean'][column]) / metadata['numeric_std'][column]\n        numeric_columns.append(values.to_numpy(dtype='float32'))\n    numeric = np.stack(numeric_columns, axis=1).astype('float32')\n\n    categorical_columns = []\n    for column in metadata['categorical_features']:\n        mapping = metadata['category_maps'][column]\n        values = frame[column].fillna('UNKNOWN').astype(str).map(mapping).fillna(0).astype('int64')\n        categorical_columns.append(values.to_numpy(dtype='int64'))\n    categorical = np.stack(categorical_columns, axis=1).astype('int64')\n    return numeric, categorical", "id": "cell-04"}, {"cell_type": "markdown", "metadata": {}, "source": "## Load data, checkpoints, and visual profiles\n\nThis cell loads full H&M data, the embedding cache, and the three final recommender checkpoints. It then builds customer visual profile sums used by image-history and late-fusion models.", "id": "cell-05"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "def normalize_article_id_series(series):\n    return series.astype(str).str.replace(r'\\.0$', '', regex=True).str.zfill(10)\n\nlog('Loading raw H&M tables...')\ntransactions = pd.read_csv(\n    TRANSACTIONS_PATH,\n    dtype={'customer_id': 'string', 'article_id': 'string'},\n    usecols=['customer_id', 'article_id', 't_dat'],\n)\ntransactions['article_id'] = normalize_article_id_series(transactions['article_id'])\ntransactions['customer_id'] = transactions['customer_id'].astype(str)\ntransactions['t_dat'] = pd.to_datetime(transactions['t_dat'])\n\ncustomers = pd.read_csv(CUSTOMERS_PATH, dtype={'customer_id': 'string'})\ncustomers['customer_id'] = customers['customer_id'].astype(str)\narticles = pd.read_csv(ARTICLES_PATH, dtype={'article_id': 'string'})\narticles['article_id'] = normalize_article_id_series(articles['article_id'])\nsample_submission = pd.read_csv(SAMPLE_SUBMISSION_PATH, dtype={'customer_id': 'string'})\nsample_submission['customer_id'] = sample_submission['customer_id'].astype(str)\n\nif SMOKE_RUN:\n    sample_submission = sample_submission.head(SMOKE_CUSTOMERS).copy()\n    log(f'Smoke mode active: {len(sample_submission):,} customers will be processed.')\nelse:\n    log(f'Full mode active: {len(sample_submission):,} customers will be processed.')\n\nlog('Loading embedding cache...')\nembeddings = np.load(EMBEDDINGS_PATH, mmap_mode=None).astype('float32')\nembedding_ids = pd.read_csv(EMBEDDING_IDS_PATH, dtype={'article_id': 'string'})\narticle_ids = normalize_article_id_series(embedding_ids['article_id']).tolist()\narticle_to_index = {article_id: index for index, article_id in enumerate(article_ids)}\nembedded_article_set = set(article_to_index)\n\nlog('Building customer visual profile arrays...')\nembedded_history = transactions[transactions['article_id'].isin(embedded_article_set)][['customer_id', 'article_id']].copy()\ncustomer_ids_with_profile = embedded_history['customer_id'].drop_duplicates().tolist()\ncustomer_to_index = {customer_id: index for index, customer_id in enumerate(customer_ids_with_profile)}\ncustomer_indices = embedded_history['customer_id'].map(customer_to_index).to_numpy(dtype='int64')\narticle_indices = embedded_history['article_id'].map(article_to_index).to_numpy(dtype='int64')\nprofile_sums = np.zeros((len(customer_ids_with_profile), embeddings.shape[1]), dtype='float32')\nfor start in tqdm(range(0, len(article_indices), PROFILE_BUILD_BATCH_SIZE), desc='Profile sum batches'):\n    end = min(start + PROFILE_BUILD_BATCH_SIZE, len(article_indices))\n    np.add.at(profile_sums, customer_indices[start:end], embeddings[article_indices[start:end]])\nprofile_counts = np.bincount(customer_indices, minlength=len(customer_ids_with_profile)).astype('float32')\ndel customer_indices, article_indices, embedded_history\ngc.collect()\n\nlog('Loading checkpoints...')\nmodels = {}\nfor name, path in MODEL_PATHS.items():\n    checkpoint, model = load_checkpoint_model(path, DEVICE)\n    models[name] = (checkpoint, model)\n    log(f'Model loaded: {name} -> {path.name}')\n\nlog('Moving lookup tensors to GPU if available...')\nARTICLE_TENSOR = torch.tensor(embeddings, dtype=torch.float32, device=DEVICE) if DEVICE.type == 'cuda' else None\nPROFILE_SUM_TENSOR = torch.tensor(profile_sums, dtype=torch.float32, device=DEVICE) if DEVICE.type == 'cuda' else None\nPROFILE_COUNT_TENSOR = torch.tensor(profile_counts, dtype=torch.float32, device=DEVICE) if DEVICE.type == 'cuda' else None\nif DEVICE.type == 'cuda':\n    allocated = torch.cuda.memory_allocated(0) / 1024**3\n    reserved = torch.cuda.memory_reserved(0) / 1024**3\n    log(f'GPU lookup tensors ready | allocated={allocated:.2f}GB | reserved={reserved:.2f}GB')\n\nprint('transactions:', transactions.shape)\nprint('customers:', customers.shape)\nprint('articles:', articles.shape)\nprint('sample_submission:', sample_submission.shape)\nprint('embeddings:', embeddings.shape)\nprint('customer profiles:', profile_sums.shape)", "id": "cell-06"}, {"cell_type": "markdown", "metadata": {}, "source": "## Candidate generation\n\nThis cell creates compact candidate pools: recent time-aware popularity, customer recent purchases, and metadata-similar items by product type and garment group.", "id": "cell-07"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "def top_articles_for_window(transactions_frame, start_date, limit, article_filter=None):\n    frame = transactions_frame\n    if start_date is not None:\n        frame = frame[frame['t_dat'] >= start_date]\n    if article_filter is not None:\n        frame = frame[frame['article_id'].isin(article_filter)]\n    return frame['article_id'].value_counts().head(limit).index.astype(str).tolist()\n\ndef build_popular_candidates(transactions_frame, article_filter):\n    log('Building time-aware popularity candidates...')\n    max_date = transactions_frame['t_dat'].max()\n    last_7d_start = max_date - pd.Timedelta(days=7)\n    last_30d_start = max_date - pd.Timedelta(days=30)\n    last_7d = top_articles_for_window(transactions_frame, last_7d_start, LAST_7D_CANDIDATES, article_filter)\n    last_30d = top_articles_for_window(transactions_frame, last_30d_start, LAST_30D_CANDIDATES, article_filter)\n    global_popular = top_articles_for_window(transactions_frame, None, GLOBAL_CANDIDATES, article_filter)\n    candidates = list(dict.fromkeys(last_7d + last_30d + global_popular))[:POPULAR_CANDIDATES]\n    log(f'Last-7d candidates: {len(last_7d):,}')\n    log(f'Last-30d candidates: {len(last_30d):,}')\n    log(f'Global candidates: {len(global_popular):,}')\n    log(f'Blended popular candidates: {len(candidates):,}')\n    return candidates\n\ndef build_customer_recent_candidates(transactions_frame, article_filter):\n    log('Building customer recent-purchase candidates...')\n    recent_transactions = transactions_frame[transactions_frame['article_id'].isin(article_filter)].copy()\n    recent_transactions = recent_transactions.sort_values(['customer_id', 't_dat'], ascending=[True, False])\n    recent_candidates = (\n        recent_transactions.groupby('customer_id')['article_id']\n        .apply(lambda values: list(dict.fromkeys(values.tolist()))[:RECENT_PURCHASE_CANDIDATES])\n        .to_dict()\n    )\n    log(f'Customers with recent candidates: {len(recent_candidates):,}')\n    return recent_candidates\n\ndef build_metadata_similarity_candidates(transactions_frame, articles_frame, article_filter):\n    log('Building metadata-similar candidates...')\n    embedded_articles_df = articles_frame[articles_frame['article_id'].isin(article_filter)].copy()\n    embedded_transactions = transactions_frame[transactions_frame['article_id'].isin(article_filter)].copy()\n    article_popularity = embedded_transactions['article_id'].value_counts()\n    embedded_articles_df['popularity'] = embedded_articles_df['article_id'].map(article_popularity).fillna(0)\n\n    product_type_top = (\n        embedded_articles_df.sort_values(['product_type_no', 'popularity'], ascending=[True, False])\n        .groupby('product_type_no')['article_id']\n        .apply(lambda values: values.head(SIMILAR_PRODUCT_TYPE_CANDIDATES).tolist())\n        .to_dict()\n    )\n    garment_group_top = (\n        embedded_articles_df.sort_values(['garment_group_no', 'popularity'], ascending=[True, False])\n        .groupby('garment_group_no')['article_id']\n        .apply(lambda values: values.head(SIMILAR_GARMENT_GROUP_CANDIDATES).tolist())\n        .to_dict()\n    )\n    article_meta = embedded_articles_df.set_index('article_id')[['product_type_no', 'garment_group_no']].to_dict('index')\n    recent_transactions = embedded_transactions.sort_values(['customer_id', 't_dat'], ascending=[True, False])\n\n    customer_candidates = {}\n    for customer_id, values in tqdm(recent_transactions.groupby('customer_id')['article_id'], desc='Metadata candidates'):\n        candidates = []\n        for article_id in list(dict.fromkeys(values.tolist()))[:RECENT_PURCHASE_CANDIDATES]:\n            meta = article_meta.get(article_id)\n            if not meta:\n                continue\n            candidates.extend(product_type_top.get(meta['product_type_no'], []))\n            candidates.extend(garment_group_top.get(meta['garment_group_no'], []))\n            if len(candidates) >= SIMILAR_PRODUCT_TYPE_CANDIDATES + SIMILAR_GARMENT_GROUP_CANDIDATES:\n                break\n        customer_candidates[customer_id] = list(dict.fromkeys(candidates))\n    log(f'Customers with metadata-similar candidates: {len(customer_candidates):,}')\n    return customer_candidates\n\npopular_articles = build_popular_candidates(transactions, embedded_article_set)\ncustomer_recent_candidates = build_customer_recent_candidates(transactions, embedded_article_set)\ncustomer_similar_candidates = build_metadata_similarity_candidates(transactions, articles, embedded_article_set)\nlog('Candidate pools ready.')", "id": "cell-08"}, {"cell_type": "markdown", "metadata": {}, "source": "## Batched scoring helpers\n\nThese helpers build a candidate grid per customer batch, compute visual scalar features, and score the same candidate grid with each model.", "id": "cell-09"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "def make_visual_batch_from_indices(article_idx, customer_idx, pair_counts):\n    if DEVICE.type == 'cuda':\n        article_b = torch.as_tensor(article_idx, dtype=torch.long, device=DEVICE)\n        customer_b = torch.as_tensor(customer_idx, dtype=torch.long, device=DEVICE)\n        pair_b = torch.as_tensor(pair_counts, dtype=torch.float32, device=DEVICE)\n        article_emb = ARTICLE_TENSOR[article_b]\n        sums = PROFILE_SUM_TENSOR[customer_b] - pair_b.unsqueeze(1) * article_emb\n        counts = PROFILE_COUNT_TENSOR[customer_b] - pair_b\n        profiles = sums / torch.clamp(counts, min=1).unsqueeze(1)\n        profiles = torch.nn.functional.normalize(profiles, p=2, dim=1)\n        profiles = torch.where(counts.unsqueeze(1) > 0, profiles, torch.zeros_like(profiles))\n        visual_similarity = torch.sum(profiles * article_emb, dim=1)\n        return article_emb, profiles, visual_similarity, counts\n\n    article_emb_np = embeddings[article_idx]\n    sums_np = profile_sums[customer_idx] - pair_counts[:, None] * article_emb_np\n    counts_np = profile_counts[customer_idx] - pair_counts\n    profiles_np = sums_np / np.maximum(counts_np, 1.0)[:, None]\n    norms = np.linalg.norm(profiles_np, axis=1, keepdims=True)\n    profiles_np = profiles_np / np.maximum(norms, 1e-8)\n    profiles_np[counts_np <= 0] = 0\n    visual_similarity_np = np.sum(profiles_np * article_emb_np, axis=1).astype('float32')\n    return (\n        torch.tensor(article_emb_np, dtype=torch.float32, device=DEVICE),\n        torch.tensor(profiles_np, dtype=torch.float32, device=DEVICE),\n        torch.tensor(visual_similarity_np, dtype=torch.float32, device=DEVICE),\n        torch.tensor(counts_np, dtype=torch.float32, device=DEVICE),\n    )\n\ndef add_visual_scalar_features(grid):\n    article_idx = grid['article_index'].to_numpy(dtype='int64')\n    customer_idx = grid['customer_index'].to_numpy(dtype='int64')\n    pair_counts = grid['pair_purchase_count'].to_numpy(dtype='float32')\n    visual_similarity = np.zeros(len(grid), dtype='float32')\n    visual_history_count = np.zeros(len(grid), dtype='float32')\n    with torch.no_grad():\n        for start in range(0, len(grid), VISUAL_FEATURE_BATCH_SIZE):\n            end = min(start + VISUAL_FEATURE_BATCH_SIZE, len(grid))\n            _, _, sim, hist = make_visual_batch_from_indices(article_idx[start:end], customer_idx[start:end], pair_counts[start:end])\n            visual_similarity[start:end] = sim.detach().cpu().numpy()\n            visual_history_count[start:end] = hist.detach().cpu().numpy()\n    grid['visual_similarity'] = visual_similarity\n    grid['visual_history_count'] = visual_history_count\n    return grid\n\ndef score_model_for_grid(model_name, checkpoint, model, grid):\n    article_idx = grid['article_index'].to_numpy(dtype='int64')\n    customer_idx = grid['customer_index'].to_numpy(dtype='int64')\n    pair_counts = grid['pair_purchase_count'].to_numpy(dtype='float32')\n\n    if model_name in {'tabular_only', 'late_fusion'}:\n        numeric, categorical = encode_tabular(grid, checkpoint['metadata'])\n    else:\n        numeric = categorical = None\n\n    scores = []\n    model.eval()\n    with torch.no_grad():\n        for start in range(0, len(grid), INFERENCE_SCORE_BATCH_SIZE):\n            end = min(start + INFERENCE_SCORE_BATCH_SIZE, len(grid))\n            if model_name == 'tabular_only':\n                numeric_b = torch.tensor(numeric[start:end], dtype=torch.float32, device=DEVICE)\n                categorical_b = torch.tensor(categorical[start:end], dtype=torch.long, device=DEVICE)\n                logits = model(numeric_b, categorical_b)\n            elif model_name == 'image_history':\n                article_emb, profiles, sim, hist = make_visual_batch_from_indices(\n                    article_idx[start:end], customer_idx[start:end], pair_counts[start:end]\n                )\n                logits = model(article_emb, profiles, sim, hist)\n            elif model_name == 'late_fusion':\n                numeric_b = torch.tensor(numeric[start:end], dtype=torch.float32, device=DEVICE)\n                categorical_b = torch.tensor(categorical[start:end], dtype=torch.long, device=DEVICE)\n                article_emb, profiles, sim, hist = make_visual_batch_from_indices(\n                    article_idx[start:end], customer_idx[start:end], pair_counts[start:end]\n                )\n                logits = model(numeric_b, categorical_b, article_emb, profiles, sim, hist)\n            else:\n                raise ValueError(model_name)\n            scores.append(torch.sigmoid(logits).detach().cpu().numpy())\n    return np.concatenate(scores).astype('float32') if scores else np.array([], dtype='float32')\n\ndef top_predictions_from_scores(grid, scores):\n    scored = grid[['customer_id', 'article_id']].copy()\n    scored['score'] = scores\n    top = (\n        scored.sort_values(['customer_id', 'score'], ascending=[True, False])\n        .groupby('customer_id')['article_id']\n        .apply(lambda values: ' '.join(list(values.head(TOP_K))))\n    )\n    return top.to_dict()\n\ndef pad_prediction(prediction, fallback_items):\n    items = str(prediction).split() if prediction else []\n    seen = set(items)\n    for item in fallback_items:\n        if len(items) >= TOP_K:\n            break\n        if item not in seen:\n            seen.add(item)\n            items.append(item)\n    return ' '.join(items[:TOP_K])", "id": "cell-10"}, {"cell_type": "markdown", "metadata": {}, "source": "## Create full submission files\n\nThis cell keeps all customers from `sample_submission.csv`, fills cold-start customers with popularity fallback, and replaces predictions for scoreable customers after model reranking.", "id": "cell-11"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "fallback_prediction = ' '.join(popular_articles[:TOP_K])\nsubmission_customers = sample_submission['customer_id'].astype(str).tolist()\ncustomer_position = {customer_id: index for index, customer_id in enumerate(submission_customers)}\nscoreable_customers = [customer_id for customer_id in submission_customers if customer_id in customer_to_index]\nscoreable_candidates = [article_id for article_id in popular_articles if article_id in article_to_index]\n\nlog(f'Submission customers: {len(submission_customers):,}')\nlog(f'Scoreable customers: {len(scoreable_customers):,}')\nlog(f'Fallback customers: {len(submission_customers) - len(scoreable_customers):,}')\nlog(f'Global candidate articles: {len(scoreable_candidates):,}')\n\nlog('Preparing shared article/customer metadata for scoring...')\nall_candidate_articles = set(scoreable_candidates)\nfor customer_id in tqdm(scoreable_customers, desc='Collecting all candidates'):\n    all_candidate_articles.update(customer_recent_candidates.get(customer_id, []))\n    all_candidate_articles.update(customer_similar_candidates.get(customer_id, []))\nall_candidate_articles = [article_id for article_id in all_candidate_articles if article_id in article_to_index]\nlog(f'Total unique candidate articles for scoring: {len(all_candidate_articles):,}')\n\narticles_meta = articles[articles['article_id'].isin(all_candidate_articles)].copy()\ncustomer_meta = customers[customers['customer_id'].isin(scoreable_customers)].copy()\npair_counts = (\n    transactions[\n        transactions['customer_id'].isin(scoreable_customers)\n        & transactions['article_id'].isin(all_candidate_articles)\n    ]\n    .groupby(['customer_id', 'article_id'])\n    .size()\n    .reset_index(name='pair_purchase_count')\n)\nlog(f'Historical customer/article pair counts: {len(pair_counts):,}')\n\nprediction_arrays = {name: np.full(len(submission_customers), fallback_prediction, dtype=object) for name in MODEL_NAMES}\nscored_customer_counts = {name: 0 for name in MODEL_NAMES}\ncandidate_counts = []\n\nfor batch_id, start in enumerate(tqdm(range(0, len(scoreable_customers), INFERENCE_CUSTOMER_BATCH_SIZE), desc='Submission customer batches'), start=1):\n    batch_customers = scoreable_customers[start:start + INFERENCE_CUSTOMER_BATCH_SIZE]\n    rows = []\n    for customer_id in batch_customers:\n        customer_candidates = []\n        customer_candidates.extend(customer_recent_candidates.get(customer_id, []))\n        customer_candidates.extend(customer_similar_candidates.get(customer_id, []))\n        customer_candidates.extend(scoreable_candidates)\n        customer_candidates = [\n            article_id for article_id in dict.fromkeys(customer_candidates)\n            if article_id in article_to_index\n        ][:CUSTOMER_CANDIDATE_LIMIT]\n        if len(customer_candidates) < TOP_K:\n            customer_candidates = list(dict.fromkeys(customer_candidates + scoreable_candidates))[:CUSTOMER_CANDIDATE_LIMIT]\n        candidate_counts.append(len(customer_candidates))\n        for article_id in customer_candidates:\n            rows.append((customer_id, article_id))\n\n    if not rows:\n        continue\n\n    grid = pd.DataFrame(rows, columns=['customer_id', 'article_id'])\n    grid = grid.merge(customer_meta, on='customer_id', how='left')\n    grid = grid.merge(articles_meta, on='article_id', how='left')\n    grid = grid.merge(pair_counts, on=['customer_id', 'article_id'], how='left')\n    grid['pair_purchase_count'] = grid['pair_purchase_count'].fillna(0).astype('float32')\n    grid['article_index'] = grid['article_id'].map(article_to_index).astype('int64')\n    grid['customer_index'] = grid['customer_id'].map(customer_to_index).astype('int64')\n    grid = add_visual_scalar_features(grid)\n\n    for model_name in MODEL_NAMES:\n        checkpoint, model = models[model_name]\n        scores = score_model_for_grid(model_name, checkpoint, model, grid)\n        top_predictions = top_predictions_from_scores(grid, scores)\n        for customer_id, prediction in top_predictions.items():\n            prediction_arrays[model_name][customer_position[customer_id]] = pad_prediction(prediction, scoreable_candidates)\n            scored_customer_counts[model_name] += 1\n\n    if batch_id % 10 == 0 or start + len(batch_customers) >= len(scoreable_customers):\n        elapsed_hours = (time.time() - START_TIME) / 3600\n        log(f'Batch {batch_id:,} complete | customers={start + len(batch_customers):,}/{len(scoreable_customers):,} | elapsed={elapsed_hours:.2f} h')\n\n    del grid, rows\n    gc.collect()\n\nsummary_lines = [\n    '# Submission Generation Summary',\n    '',\n    f'- Smoke run: {SMOKE_RUN}',\n    f'- Customers in output: {len(submission_customers):,}',\n    f'- Scoreable customers: {len(scoreable_customers):,}',\n    f'- Fallback customers before scoring: {len(submission_customers) - len(scoreable_customers):,}',\n    f'- Top-k: {TOP_K}',\n    f'- Customer candidate limit: {CUSTOMER_CANDIDATE_LIMIT}',\n    f'- Inference customer batch size: {INFERENCE_CUSTOMER_BATCH_SIZE}',\n    f'- Inference score batch size: {INFERENCE_SCORE_BATCH_SIZE}',\n    f'- Mean candidate count: {float(np.mean(candidate_counts)) if candidate_counts else 0:.2f}',\n    f'- Runtime hours: {(time.time() - START_TIME) / 3600:.2f}',\n    f'- Device: {DEVICE}',\n    '',\n    '| model | output | rows | scored_customers | fallback_rows | unique_predictions | valid_top12 |',\n    '| --- | --- | ---: | ---: | ---: | ---: | --- |',\n]\n\nfor model_name in MODEL_NAMES:\n    output = pd.DataFrame({'customer_id': submission_customers, 'prediction': prediction_arrays[model_name]})\n    output['prediction'] = output['prediction'].map(lambda value: pad_prediction(value, scoreable_candidates))\n    output_path = WORK_DIR / f'submission_{model_name}.csv'\n    output.to_csv(output_path, index=False)\n    prediction_lengths = output['prediction'].astype(str).str.split().map(len)\n    valid_top12 = bool(prediction_lengths.eq(TOP_K).all())\n    if list(output.columns) != ['customer_id', 'prediction']:\n        raise AssertionError(f'{model_name}: invalid columns')\n    if len(output) != len(sample_submission):\n        raise AssertionError(f'{model_name}: row count does not match sample submission')\n    if not valid_top12:\n        raise AssertionError(f'{model_name}: some predictions do not contain {TOP_K} articles')\n    unique_predictions = int(output['prediction'].nunique())\n    fallback_rows = int((output['prediction'] == fallback_prediction).sum())\n    summary_lines.append(\n        f'| {model_name} | `{output_path.name}` | {len(output):,} | {scored_customer_counts[model_name]:,} | {fallback_rows:,} | {unique_predictions:,} | {valid_top12} |'\n    )\n    log(f'Saved: {output_path}')\n\nsummary_path = WORK_DIR / 'submission_generation_summary.md'\nsummary_path.write_text('\\n'.join(summary_lines) + '\\n', encoding='utf-8')\nprint('\\n'.join(summary_lines))\nlog(f'Summary saved: {summary_path}')", "id": "cell-12"}, {"cell_type": "markdown", "metadata": {}, "source": "## Submit commands\n\nDo not submit automatically from this notebook. After full output validation, submit manually in this order:\n\n```bash\nkaggle competitions submit -c h-and-m-personalized-fashion-recommendations -f /kaggle/working/submission_late_fusion.csv -m \"perseptron v2 late_fusion optimized full\"\nkaggle competitions submit -c h-and-m-personalized-fashion-recommendations -f /kaggle/working/submission_tabular_only.csv -m \"perseptron v2 tabular_only optimized full\"\nkaggle competitions submit -c h-and-m-personalized-fashion-recommendations -f /kaggle/working/submission_image_history.csv -m \"perseptron v2 image_history optimized full\"\n```", "id": "cell-13"}], "metadata": {"kernelspec": {"display_name": "Python 3", "language": "python", "name": "python3"}, "language_info": {"name": "python", "pygments_lexer": "ipython3"}}, "nbformat": 4, "nbformat_minor": 5}