{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# H&M development sample\nCreates a leakage-safe, customer-based sample from the H&M Kaggle data. Eligibility is computed from training data only; complete 26-week histories are then retained for nested 500, 5,000, and 20,000-customer cohorts.\n\n```text\narticles + article_first_seen\n        │\n        ├── training-available catalog\n        │          │\n        │          ├── vocabulary-building sample\n        │          └── held-out vocabulary evaluation\n        │\n        └── assignment_articles\n                    │\n                    └── frozen vocabulary assignment\n                              │\n                              ▼\ntransactions + assigned tags + customer cohorts\n                              │\n                              ├── train recommender\n                              ├── validate recommender\n                              └── evaluate by customer/item subgroup\n```","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:21:57.972932Z","iopub.execute_input":"2026-08-21T19:21:57.973246Z","iopub.status.idle":"2026-08-21T19:21:58.234423Z","shell.execute_reply.started":"2026-08-21T19:21:57.973189Z","shell.execute_reply":"2026-08-21T19:21:58.233640Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Read Data","metadata":{}},{"cell_type":"code","source":"art_dtypes = {'article_id': 'string'}\ncust_dtypes = {'customer_id': 'string'}\ntx_dtypes = {\n    'customer_id': 'string',\n    'article_id': 'uint32',\n    'price': 'float32',\n    'sales_channel_id': 'uint8',\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:22:03.902305Z","iopub.execute_input":"2026-08-21T19:22:03.902720Z","iopub.status.idle":"2026-08-21T19:22:03.907910Z","shell.execute_reply.started":"2026-08-21T19:22:03.902688Z","shell.execute_reply":"2026-08-21T19:22:03.907172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles = pd.read_csv('/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/articles.csv',\n                       dtype=art_dtypes)\narticles['article_id'] = articles['article_id'].str.zfill(10)\ncustomers = pd.read_csv('/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/customers.csv',\n                        dtype=cust_dtypes)\ntransactions = pd.read_csv('/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/transactions_train.csv',\n                           dtype=tx_dtypes,\n                           usecols=['t_dat', 'customer_id', 'article_id', 'sales_channel_id'])\nprint(f\"Number of articles: {articles.shape[0]}\")\nprint(f\"Number of customers: {customers.shape[0]}\")\nprint(f\"Number of transactions: {transactions.shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:22:06.163665Z","iopub.execute_input":"2026-08-21T19:22:06.164018Z","iopub.status.idle":"2026-08-21T19:22:43.797858Z","shell.execute_reply.started":"2026-08-21T19:22:06.163935Z","shell.execute_reply":"2026-08-21T19:22:43.796996Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Build training-only customer eligibility\nA purchase event is a distinct (customer, date, article) tuple. This prevents duplicate quantity rows from making a one-basket customer appear to have a long history.","metadata":{}},{"cell_type":"code","source":"HISTORY_START = '2020-03-11'\nTRAIN_END = '2020-09-08'\nVALID_START, VALID_END = '2020-09-09', '2020-09-15'\nTEST_START, TEST_END = '2020-09-16', '2020-09-22'\nRECENT_START = '2020-07-15'  # eight weeks before validation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:22:43.799302Z","iopub.execute_input":"2026-08-21T19:22:43.799605Z","iopub.status.idle":"2026-08-21T19:22:43.803698Z","shell.execute_reply.started":"2026-08-21T19:22:43.799584Z","shell.execute_reply":"2026-08-21T19:22:43.802919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# filter to transactions within train period\nkeep = transactions['t_dat'].between(HISTORY_START, TRAIN_END)\ntx_keep = transactions.loc[keep]\nprint(f\"Number of filtered transactions: {tx_keep.shape[0]}\")\n\n# filter down to distinct customer/article purchases on particular dates\nevents = tx_keep.drop_duplicates(['customer_id', 't_dat', 'article_id'])\nprint(f\"Number of distinct customer/date/article combinations: {events.shape[0]}\")\n\n# calculate customer eligibility criteria metrics\ncustomer_stats = events.groupby('customer_id', observed=True).agg(\n    event_count=('article_id', 'size'),\n    purchase_dates=('t_dat', 'nunique'),\n    distinct_articles=('article_id', 'nunique'),\n    last_purchase=('t_dat', 'max'),\n)\nprint(f\"Number of customers: {customer_stats.shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:22:55.404994Z","iopub.execute_input":"2026-08-21T19:22:55.405301Z","iopub.status.idle":"2026-08-21T19:23:37.308350Z","shell.execute_reply.started":"2026-08-21T19:22:55.405277Z","shell.execute_reply":"2026-08-21T19:23:37.307610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"channel_counts = events.groupby(\n    ['customer_id', 'sales_channel_id'], observed=True\n).size().rename('n').reset_index()\n\ndominant_channel = (\n    channel_counts.sort_values(\n        ['customer_id', 'n', 'sales_channel_id'],\n        ascending=[True, False, True],\n    ).drop_duplicates('customer_id').set_index('customer_id')['sales_channel_id']\n)\n\ncustomer_stats['dominant_channel'] = dominant_channel\ncustomer_stats['last_purchase'] = pd.to_datetime(customer_stats['last_purchase'])\ncustomer_stats['days_since_last'] = (\n    pd.Timestamp(TRAIN_END) - customer_stats['last_purchase']\n).dt.days\ncustomer_stats['repeat_share'] = (\n    1 - customer_stats['distinct_articles'] / customer_stats['event_count']\n).clip(0, 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:24:43.620017Z","iopub.execute_input":"2026-08-21T19:24:43.620327Z","iopub.status.idle":"2026-08-21T19:24:48.741228Z","shell.execute_reply.started":"2026-08-21T19:24:43.620306Z","shell.execute_reply":"2026-08-21T19:24:48.740324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# filter down to sampling criteria\neligible = customer_stats.loc[\n    (customer_stats['event_count'] >= 5)\n    & (customer_stats['purchase_dates'] >= 3)\n    & (customer_stats['distinct_articles'] >= 2)\n    & (customer_stats['last_purchase'] >= RECENT_START)\n].reset_index()\nprint(f\"Number of customers who are eligible: {eligible.shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:24:51.411265Z","iopub.execute_input":"2026-08-21T19:24:51.411597Z","iopub.status.idle":"2026-08-21T19:24:51.623900Z","shell.execute_reply.started":"2026-08-21T19:24:51.411572Z","shell.execute_reply":"2026-08-21T19:24:51.623037Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Deterministic stratified customer sample\n\nCustomers are stratified by activity, recency, repeat behavior, and dominant sales channel. Ranking a stable hash within each stratum preserves the eligible population mix and makes all cohort sizes nested.","metadata":{}},{"cell_type":"code","source":"SEED=2026\nCOHORT_SIZES = (500, 5_000, 20_000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:24:56.330738Z","iopub.execute_input":"2026-08-21T19:24:56.331010Z","iopub.status.idle":"2026-08-21T19:24:56.335310Z","shell.execute_reply.started":"2026-08-21T19:24:56.330988Z","shell.execute_reply":"2026-08-21T19:24:56.334598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eligible['activity_band'] = pd.cut(\n    eligible['event_count'],\n    bins=[4, 7, 15, 30, np.inf],\n    labels=['5-7', '8-15', '16-30', '31+'],\n)\neligible['recency_band'] = pd.cut(\n    eligible['days_since_last'],\n    bins=[-1, 7, 28, 56],\n    labels=['0-7d', '8-28d', '29-56d'],\n)\neligible['repeat_band'] = pd.cut(\n    eligible['repeat_share'],\n    bins=[-0.001, 0.25, 0.50, 0.75, 1.001],\n    labels=['low', 'medium', 'high', 'very_high'],\n)\nstratum_cols = ['activity_band', 'recency_band', 'dominant_channel']\neligible['stratum'] = eligible[stratum_cols].astype('string').agg('|'.join, axis=1)\neligible['sample_hash'] = pd.util.hash_pandas_object(\n    eligible['customer_id'] + f'|{SEED}', index=False\n).astype('uint64')\neligible['within_stratum_percentile'] = eligible.groupby('stratum')[\n    'sample_hash'\n].rank(method='first', pct=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:25:01.289704Z","iopub.execute_input":"2026-08-21T19:25:01.290031Z","iopub.status.idle":"2026-08-21T19:25:07.990962Z","shell.execute_reply.started":"2026-08-21T19:25:01.290008Z","shell.execute_reply":"2026-08-21T19:25:07.990059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected = (\n    eligible.sort_values(['within_stratum_percentile', 'sample_hash'])\n            .head(max(COHORT_SIZES))\n            .reset_index(drop=True)\n)\nselected['sample_order'] = np.arange(1, len(selected) + 1)\nselected['smallest_cohort'] = np.select(\n    [selected['sample_order'] <= size for size in COHORT_SIZES[:-1]],\n    COHORT_SIZES[:-1],\n    default=COHORT_SIZES[-1],\n).astype('int32')\nselected_ids = set(selected['customer_id'])\neligible_count = int(len(eligible))\n\nprint(selected['smallest_cohort'].value_counts().sort_index())\npd.crosstab(selected['activity_band'], selected['recency_band'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:25:13.434909Z","iopub.execute_input":"2026-08-21T19:25:13.435181Z","iopub.status.idle":"2026-08-21T19:25:13.606467Z","shell.execute_reply.started":"2026-08-21T19:25:13.435158Z","shell.execute_reply":"2026-08-21T19:25:13.605346Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Extract complete sampled histories\n\nThis second pass keeps every row—including legitimate duplicate quantities—for selected customers during the 26-week window. It also records each article's first appearance in the full transaction file.","metadata":{}},{"cell_type":"code","source":"keep = (\n        transactions['t_dat'].between(HISTORY_START, TEST_END)\n        & transactions['customer_id'].isin(selected_ids)\n    )\nsampled_tx = transactions.loc[keep].reset_index(drop=True)\nprint(f\"Full transaction sample: {sampled_tx.shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:25:27.149095Z","iopub.execute_input":"2026-08-21T19:25:27.150127Z","iopub.status.idle":"2026-08-21T19:25:32.197285Z","shell.execute_reply.started":"2026-08-21T19:25:27.150082Z","shell.execute_reply":"2026-08-21T19:25:32.196663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_tx['split'] = np.select(\n    [\n        sampled_tx['t_dat'] <= TRAIN_END,\n        sampled_tx['t_dat'].between(VALID_START, VALID_END),\n        sampled_tx['t_dat'].between(TEST_START, TEST_END),\n    ],\n    ['train', 'validation', 'test'],\n    default='outside',\n)\nsampled_tx['split'].value_counts(normalize=True)\n\nsampled_tx = sampled_tx.loc[sampled_tx['split'] != 'outside'].copy()\nsampled_tx['article_id'] = sampled_tx['article_id'].astype('string').str.zfill(10)\nsampled_tx['split'] = sampled_tx['split'].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:25:43.021989Z","iopub.execute_input":"2026-08-21T19:25:43.022392Z","iopub.status.idle":"2026-08-21T19:25:43.507040Z","shell.execute_reply.started":"2026-08-21T19:25:43.022364Z","shell.execute_reply":"2026-08-21T19:25:43.505886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"first_seen_parts = transactions.groupby('article_id', observed=True)['t_dat'].min()\nprint(f\"Number of first seen articles: {first_seen_parts.shape[0]}\")\n\narticle_first_seen = (\n    first_seen_parts.groupby(level=0).min()\n      .rename('first_seen')\n      .reset_index()\n)\narticle_first_seen['article_id'] = (\n    article_first_seen['article_id'].astype('string').str.zfill(10)\n)\n\nprint(f'{len(sampled_tx):,} sampled transaction rows')\nprint(f\"{sampled_tx['customer_id'].nunique():,} customers\")\nprint(f\"{sampled_tx['article_id'].nunique():,} articles\")\nsampled_tx.groupby('split', observed=True).agg(\n    rows=('article_id', 'size'),\n    customers=('customer_id', 'nunique'),\n    articles=('article_id', 'nunique'),\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:25:59.383239Z","iopub.execute_input":"2026-08-21T19:25:59.383579Z","iopub.status.idle":"2026-08-21T19:26:12.099327Z","shell.execute_reply.started":"2026-08-21T19:25:59.383555Z","shell.execute_reply":"2026-08-21T19:26:12.098336Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Export the portable sample","metadata":{}},{"cell_type":"code","source":"sampled_article_ids = set(sampled_tx['article_id'])\nassignment_articles = articles.loc[articles['article_id'].isin(sampled_article_ids)].copy()\nprint(f\"Number of sampled articles: {assignment_articles.shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:26:21.837048Z","iopub.execute_input":"2026-08-21T19:26:21.837402Z","iopub.status.idle":"2026-08-21T19:26:22.074043Z","shell.execute_reply.started":"2026-08-21T19:26:21.837375Z","shell.execute_reply":"2026-08-21T19:26:22.073263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_customers = customers.loc[customers['customer_id'].isin(selected_ids)].copy()\nprint(f\"Number of sampled customers: {sampled_customers.shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:26:23.925067Z","iopub.execute_input":"2026-08-21T19:26:23.925427Z","iopub.status.idle":"2026-08-21T19:26:24.240721Z","shell.execute_reply.started":"2026-08-21T19:26:23.925401Z","shell.execute_reply":"2026-08-21T19:26:24.239615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cohort_columns = [\n    'customer_id', 'sample_order', 'smallest_cohort', 'stratum',\n    'event_count', 'purchase_dates', 'distinct_articles', 'last_purchase',\n    'days_since_last', 'repeat_share', 'dominant_channel',\n]\ncustomer_cohorts = selected[cohort_columns].copy()\nprint(f\"Number of customers in cohort dataframe: {customer_cohorts.shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:26:26.823042Z","iopub.execute_input":"2026-08-21T19:26:26.823374Z","iopub.status.idle":"2026-08-21T19:26:26.835878Z","shell.execute_reply.started":"2026-08-21T19:26:26.823350Z","shell.execute_reply":"2026-08-21T19:26:26.834946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sampled_tx.to_parquet('transactions.parquet', index=False)\narticles.to_parquet('articles.parquet', index=False)\nassignment_articles.to_parquet('assignment_articles.parquet', index=False)\nsampled_customers.to_parquet('customers.parquet', index=False)\ncustomer_cohorts.to_parquet('customer_cohorts.parquet', index=False)\narticle_first_seen.to_parquet('article_first_seen.parquet', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T19:26:40.339289Z","iopub.execute_input":"2026-08-21T19:26:40.339600Z","iopub.status.idle":"2026-08-21T19:26:41.179343Z","shell.execute_reply.started":"2026-08-21T19:26:40.339575Z","shell.execute_reply":"2026-08-21T19:26:41.178250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}