{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The action items are:\n- Understand the competition H&M\n- Create an initial plan\n- Implement the initial plan and run on either Kaggle or VM","metadata":{}},{"cell_type":"code","source":"# \"\"\"\n# H&M Personalized Fashion Recommendations - Baseline Implementation\n# Following the baseline plan for MLE-Bench competition\n# Expected MAP@12: 0.012-0.015\n# Runtime: 30-60 minutes\n# \"\"\"\n\n# import pandas as pd\n# import numpy as np\n# from datetime import datetime, timedelta\n# from collections import defaultdict\n# import warnings\n# warnings.filterwarnings('ignore')\n\n# def reduce_mem_usage(df):\n#     \"\"\"\n#     Reduce memory usage by downcasting numeric dtypes.\n#     This is critical for handling the large H&M dataset.\n#     \"\"\"\n#     start_mem = df.memory_usage().sum() / 1024**2\n#     print(f'Memory usage: {start_mem:.2f} MB')\n\n#     for col in df.columns:\n#         col_type = df[col].dtype\n\n#         if col_type != object:\n#             c_min = df[col].min()\n#             c_max = df[col].max()\n\n#             if str(col_type)[:3] == 'int':\n#                 if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n#                     df[col] = df[col].astype(np.int8)\n#                 elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n#                     df[col] = df[col].astype(np.int16)\n#                 elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n#                     df[col] = df[col].astype(np.int32)\n#                 elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n#                     df[col] = df[col].astype(np.int64)\n#             else:\n#                 if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n#                     df[col] = df[col].astype(np.float32)\n#                 elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n#                     df[col] = df[col].astype(np.float32)\n#                 else:\n#                     df[col] = df[col].astype(np.float64)\n\n#     end_mem = df.memory_usage().sum() / 1024**2\n#     print(f'Memory usage after optimization: {end_mem:.2f} MB')\n#     print(f'Decreased by {100 * (start_mem - end_mem) / start_mem:.1f}%')\n\n#     return df\n\n\n# def calculate_popular_items(transactions_df, cutoff_date, n=20):\n#     \"\"\"\n#     Calculate the most popular items in the last 7 days before cutoff_date.\n#     These serve as fallback recommendations for customers with limited history.\n#     \"\"\"\n#     last_week = cutoff_date - timedelta(days=7)\n#     recent_transactions = transactions_df[transactions_df['t_dat'] >= last_week].copy()\n\n#     popular_items = (recent_transactions['article_id']\n#                     .value_counts()\n#                     .head(n)\n#                     .index.tolist())\n\n#     print(f\"Calculated {len(popular_items)} popular items from last 7 days\")\n#     return popular_items\n\n\n# def get_customer_recommendations(customer_id, transactions_df, cutoff_date, \n#                                  popular_items, n=12, lookback_days=30):\n#     \"\"\"\n#     Generate recommendations for a single customer using time-weighted purchase history.\n\n#     Strategy:\n#     1. Extract customer's purchases from last 30 days before cutoff_date\n#     2. Apply time decay weighting: weight = 1 / (1 + days_ago)\n#     3. Aggregate weights by article_id\n#     4. Select top 12 by weight\n#     5. Fill gaps with popular items if < 12 recommendations\n\n#     Args:\n#         customer_id: Customer identifier\n#         transactions_df: Transactions dataframe\n#         cutoff_date: Date to predict from (e.g., validation start date)\n#         popular_items: List of popular item fallbacks\n#         n: Number of recommendations to generate (default 12)\n#         lookback_days: How far back to look for purchase history (default 30)\n\n#     Returns:\n#         List of article_ids (length = n)\n#     \"\"\"\n#     # Extract customer's recent purchase history\n#     lookback_date = cutoff_date - timedelta(days=lookback_days)\n#     customer_history = transactions_df[\n#         (transactions_df['customer_id'] == customer_id) &\n#         (transactions_df['t_dat'] >= lookback_date) &\n#         (transactions_df['t_dat'] < cutoff_date)\n#     ].copy()\n\n#     if len(customer_history) == 0:\n#         # No purchase history - return popular items\n#         return popular_items[:n]\n\n#     # Calculate time decay weights\n#     customer_history['days_ago'] = (cutoff_date - customer_history['t_dat']).dt.days\n#     customer_history['weight'] = 1.0 / (1.0 + customer_history['days_ago'])\n\n#     # Aggregate weights by article_id\n#     article_weights = (customer_history.groupby('article_id')['weight']\n#                       .sum()\n#                       .sort_values(ascending=False))\n\n#     # Get top recommendations\n#     recommendations = article_weights.head(n).index.tolist()\n\n#     # Fill with popular items if we have fewer than n recommendations\n#     if len(recommendations) < n:\n#         # Add popular items that aren't already in recommendations\n#         for item in popular_items:\n#             if item not in recommendations:\n#                 recommendations.append(item)\n#                 if len(recommendations) == n:\n#                     break\n\n#     return recommendations[:n]\n\n\n# def generate_all_recommendations(transactions_df, customer_ids, cutoff_date, \n#                                 popular_items, n=12):\n#     \"\"\"\n#     Generate recommendations for all customers.\n\n#     Args:\n#         transactions_df: Transactions dataframe\n#         customer_ids: List of customer IDs to generate predictions for\n#         cutoff_date: Date to predict from\n#         popular_items: List of popular item fallbacks\n#         n: Number of recommendations per customer\n\n#     Returns:\n#         Dictionary mapping customer_id to list of article_ids\n#     \"\"\"\n#     recommendations = {}\n#     total_customers = len(customer_ids)\n\n#     print(f\"Generating recommendations for {total_customers} customers...\")\n\n#     for idx, customer_id in enumerate(customer_ids):\n#         if (idx + 1) % 10000 == 0:\n#             print(f\"Processed {idx + 1}/{total_customers} customers\")\n\n#         recs = get_customer_recommendations(\n#             customer_id, transactions_df, cutoff_date, \n#             popular_items, n=n\n#         )\n#         recommendations[customer_id] = recs\n\n#     print(f\"Completed recommendations for all {total_customers} customers\")\n#     return recommendations\n\n\n# def create_submission(recommendations, output_file='submission.csv'):\n#     \"\"\"\n#     Create submission file in the required format.\n\n#     Format: customer_id,prediction\n#     where prediction is space-separated list of 12 article_ids\n#     \"\"\"\n#     submission_data = []\n\n#     for customer_id, article_list in recommendations.items():\n#         # Convert article_ids to strings and join with spaces\n#         prediction = ' '.join([str(article_id).zfill(10) for article_id in article_list])\n#         submission_data.append({\n#             'customer_id': customer_id,\n#             'prediction': prediction\n#         })\n\n#     submission_df = pd.DataFrame(submission_data)\n#     submission_df.to_csv(output_file, index=False)\n#     print(f\"Submission saved to {output_file}\")\n#     print(f\"Total predictions: {len(submission_df)}\")\n\n#     return submission_df\n\n\n# def calculate_map_at_k(actual, predicted, k=12):\n#     \"\"\"\n#     Calculate Mean Average Precision at K for validation.\n\n#     Args:\n#         actual: Dictionary mapping customer_id to list of actual purchased article_ids\n#         predicted: Dictionary mapping customer_id to list of predicted article_ids\n#         k: Number of recommendations (default 12)\n\n#     Returns:\n#         MAP@K score\n#     \"\"\"\n#     aps = []\n\n#     for customer_id, pred_items in predicted.items():\n#         if customer_id not in actual:\n#             continue\n\n#         actual_items = set(actual[customer_id])\n#         if len(actual_items) == 0:\n#             continue\n\n#         pred_items = pred_items[:k]\n\n#         score = 0.0\n#         num_hits = 0.0\n\n#         for i, item in enumerate(pred_items):\n#             if item in actual_items:\n#                 num_hits += 1.0\n#                 score += num_hits / (i + 1.0)\n\n#         if len(actual_items) > 0:\n#             aps.append(score / min(len(actual_items), k))\n\n#     return np.mean(aps) if len(aps) > 0 else 0.0\n\n\n# def main():\n#     \"\"\"\n#     Main execution function for H&M baseline recommendations.\n#     \"\"\"\n#     print(\"=\"*80)\n#     print(\"H&M Personalized Fashion Recommendations - Baseline\")\n#     print(\"=\"*80)\n\n#     # 1. Load Data\n#     print(\"\\n1. Loading data...\")\n#     transactions = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', parse_dates=['t_dat'])\n#     customers = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\n#     articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\n\n#     print(f\"Transactions shape: {transactions.shape}\")\n#     print(f\"Customers shape: {customers.shape}\")\n#     print(f\"Articles shape: {articles.shape}\")\n\n#     # 2. Memory Optimization\n#     print(\"\\n2. Optimizing memory usage...\")\n#     transactions = reduce_mem_usage(transactions)\n#     customers = reduce_mem_usage(customers)\n#     articles = reduce_mem_usage(articles)\n\n#     # 3. Create Validation Split\n#     print(\"\\n3. Creating validation split...\")\n#     SPLIT_DATE = pd.to_datetime('2020-09-16')\n#     VALIDATION_END = pd.to_datetime('2020-09-23')\n\n#     train_df = transactions[transactions['t_dat'] < SPLIT_DATE].copy()\n#     valid_df = transactions[\n#         (transactions['t_dat'] >= SPLIT_DATE) &\n#         (transactions['t_dat'] < VALIDATION_END)\n#     ].copy()\n\n#     print(f\"Train size: {len(train_df):,} transactions\")\n#     print(f\"Valid size: {len(valid_df):,} transactions\")\n#     print(f\"Train date range: {train_df['t_dat'].min()} to {train_df['t_dat'].max()}\")\n#     print(f\"Valid date range: {valid_df['t_dat'].min()} to {valid_df['t_dat'].max()}\")\n\n#     # 4. Calculate Popular Items\n#     print(\"\\n4. Calculating popular items...\")\n#     popular_items = calculate_popular_items(train_df, SPLIT_DATE, n=20)\n#     print(f\"Top 5 popular items: {popular_items[:5]}\")\n\n#     # 5. Generate Validation Recommendations\n#     print(\"\\n5. Generating validation recommendations...\")\n#     validation_customers = valid_df['customer_id'].unique()\n#     print(f\"Validation customers: {len(validation_customers)}\")\n\n#     validation_recommendations = generate_all_recommendations(\n#         train_df, validation_customers, SPLIT_DATE, popular_items, n=12\n#     )\n\n#     # 6. Calculate MAP@12 on Validation Set\n#     print(\"\\n6. Calculating MAP@12 on validation set...\")\n#     validation_actual = defaultdict(list)\n#     for _, row in valid_df.iterrows():\n#         validation_actual[row['customer_id']].append(row['article_id'])\n\n#     validation_map12 = calculate_map_at_k(validation_actual, validation_recommendations, k=12)\n#     print(f\"Validation MAP@12: {validation_map12:.6f}\")\n\n#     # 7. Generate Test Predictions\n#     print(\"\\n7. Generating test predictions...\")\n\n#     # For test set, use all training data up to max date\n#     TEST_CUTOFF = transactions['t_dat'].max()\n\n#     # Load or infer test customer IDs\n#     # If sample_submission.csv exists, use those customer IDs\n#     try:\n#         sample_submission = pd.read_csv('./data/sample_submission.csv')\n#         test_customers = sample_submission['customer_id'].unique()\n#         print(f\"Loaded {len(test_customers)} test customers from sample_submission.csv\")\n#     except:\n#         # Otherwise, use all unique customers from transactions\n#         test_customers = transactions['customer_id'].unique()\n#         print(f\"Using {len(test_customers)} unique customers from transactions\")\n\n#     # Recalculate popular items using all training data\n#     popular_items_test = calculate_popular_items(transactions, TEST_CUTOFF, n=20)\n\n#     test_recommendations = generate_all_recommendations(\n#         transactions, test_customers, TEST_CUTOFF, popular_items_test, n=12\n#     )\n\n#     # 8. Create Submission File\n#     print(\"\\n8. Creating submission file...\")\n#     submission_df = create_submission(test_recommendations, output_file='submission.csv')\n\n#     # 9. Summary\n#     print(\"\\n\" + \"=\"*80)\n#     print(\"BASELINE COMPLETE\")\n#     print(\"=\"*80)\n#     print(f\"Validation MAP@12: {validation_map12:.6f}\")\n#     print(f\"Submission created: submission.csv\")\n#     print(f\"Total predictions: {len(submission_df)}\")\n#     print(\"=\"*80)\n\n#     return validation_map12\n\n\n# if __name__ == \"__main__\":\n#     validation_score = main()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T18:10:08.913420Z","iopub.execute_input":"2025-11-03T18:10:08.913709Z","iopub.status.idle":"2025-11-03T18:11:33.513593Z","shell.execute_reply.started":"2025-11-03T18:10:08.913686Z","shell.execute_reply":"2025-11-03T18:11:33.512463Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-31T06:39:31.750355Z","iopub.execute_input":"2025-10-31T06:39:31.750743Z","iopub.status.idle":"2025-10-31T06:43:54.779410Z","shell.execute_reply.started":"2025-10-31T06:39:31.750717Z","shell.execute_reply":"2025-10-31T06:43:54.778231Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and explore the customers dataset\nimport pandas as pd\n\n# Read the customers CSV file\ncustomers_df = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\n\n# Display basic information\nprint(\"Customers Dataset Shape:\", customers_df.shape)\nprint(\"\\nFirst 5 rows:\")\nprint(customers_df.head())\nprint(\"\\nColumn names and types:\")\nprint(customers_df.dtypes)\nprint(\"\\nMissing values:\")\nprint(customers_df.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T07:24:24.342135Z","iopub.execute_input":"2025-10-31T07:24:24.343005Z","iopub.status.idle":"2025-10-31T07:24:32.455830Z","shell.execute_reply.started":"2025-10-31T07:24:24.342970Z","shell.execute_reply":"2025-10-31T07:24:32.454563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom datetime import datetime, timedelta\nfrom collections import defaultdict\nimport warnings\nimport gc\nimport time\nwarnings.filterwarnings('ignore')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T19:17:38.167709Z","iopub.execute_input":"2025-11-03T19:17:38.168395Z","iopub.status.idle":"2025-11-03T19:17:38.432220Z","shell.execute_reply.started":"2025-11-03T19:17:38.168369Z","shell.execute_reply":"2025-11-03T19:17:38.431386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type == object or pd.api.types.is_datetime64_any_dtype(df[col]):\n            continue\n        c_min = df[col].min()\n        c_max = df[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                df[col] = df[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                df[col] = df[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                df[col] = df[col].astype(np.int32)\n        elif str(col_type)[:5] == 'float':\n            if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                df[col] = df[col].astype(np.float32)\n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory: {start_mem:.1f}MB → {end_mem:.1f}MB')\n    return df\n\ndef memory_cleanup():\n    gc.collect()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T19:17:39.515375Z","iopub.execute_input":"2025-11-03T19:17:39.516304Z","iopub.status.idle":"2025-11-03T19:17:39.526322Z","shell.execute_reply.started":"2025-11-03T19:17:39.516268Z","shell.execute_reply":"2025-11-03T19:17:39.525558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_transactions_chunked(filepath, chunksize=500000, usecols=None, parse_dates=None):\n    print(f\"Loading {filepath}...\")\n    chunks = []\n    for i, chunk in enumerate(pd.read_csv(filepath, chunksize=chunksize, usecols=usecols, parse_dates=parse_dates)):\n        chunk = reduce_mem_usage(chunk)\n        chunks.append(chunk)\n        if (i+1) % 20 == 0:\n            print(f\"  {i+1} chunks loaded...\")\n    transactions = pd.concat(chunks, ignore_index=True)\n    del chunks\n    memory_cleanup()\n    print(f\"✓ Loaded {len(transactions):,} rows\")\n    return transactions\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T19:17:40.894056Z","iopub.execute_input":"2025-11-03T19:17:40.894338Z","iopub.status.idle":"2025-11-03T19:17:40.899538Z","shell.execute_reply.started":"2025-11-03T19:17:40.894318Z","shell.execute_reply":"2025-11-03T19:17:40.898719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compute_popular_items(df, cutoff_date, n=20):\n    cutoff = cutoff_date - timedelta(days=7)\n    recent = df[df['t_dat'] >= cutoff]\n    items = recent['article_id'].value_counts().head(n).index.values\n    del recent\n    memory_cleanup()\n    return items\n\ndef get_recommendations(customer_id, customer_index, cutoff_date, popular_items, n=12):\n    if customer_id not in customer_index:\n        return popular_items[:n].tolist()\n    \n    data = customer_index[customer_id]\n    cutoff = cutoff_date - timedelta(days=30)\n    mask = data['dates'] >= cutoff\n    \n    if not mask.any():\n        return popular_items[:n].tolist()\n    \n    recent_articles = data['article_ids'][mask]\n    recent_dates = data['dates'][mask]\n    \n    # Fixed datetime calculation\n    cutoff_ts = pd.Timestamp(cutoff_date)\n    days_ago = (cutoff_ts - pd.Series(recent_dates)).dt.days.values\n    weights = 1.0 / (1.0 + days_ago)\n    \n    df = pd.DataFrame({'article_id': recent_articles, 'weight': weights})\n    top = df.groupby('article_id')['weight'].sum().nlargest(n).index.values.tolist()\n    \n    # Fill with popular items\n    while len(top) < n:\n        for item in popular_items:\n            if item not in top:\n                top.append(item)\n                if len(top) >= n:\n                    break\n        break\n    \n    return top[:n]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T19:17:55.273140Z","iopub.execute_input":"2025-11-03T19:17:55.273732Z","iopub.status.idle":"2025-11-03T19:17:55.280835Z","shell.execute_reply.started":"2025-11-03T19:17:55.273707Z","shell.execute_reply":"2025-11-03T19:17:55.280106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\ntransactions = load_transactions_chunked(\n    '/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv',\n    chunksize=500000,\n    usecols=['customer_id', 'article_id', 't_dat'],\n    parse_dates=['t_dat']\n)\n\n# Compute popular items\ncutoff_date = transactions['t_dat'].max()\npopular_items = compute_popular_items(transactions, cutoff_date)\n\n# Build customer index\nprint(\"Building customer index...\")\ntransactions = transactions.sort_values(['customer_id', 't_dat'])\ncustomer_index = {}\nfor cust_id, group in transactions.groupby('customer_id'):\n    customer_index[cust_id] = {\n        'article_ids': group['article_id'].values,\n        'dates': group['t_dat'].values\n    }\n\n# Load customer IDs\nsample = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\ncustomer_ids = sample['customer_id'].values\n\n# Generate recommendations\nrecommendations = {}\nstart = time.time()\nfor i, cid in enumerate(customer_ids):\n    recommendations[cid] = get_recommendations(cid, customer_index, cutoff_date, popular_items)\n    if (i+1) % 10000 == 0:\n        print(f\"  {i+1:,} / {len(customer_ids):,}\")\n\n# Create submission\nsubmission = pd.DataFrame({\n    'customer_id': list(recommendations.keys()),\n    'prediction': [' '.join([str(int(x)).zfill(10) for x in v]) for v in recommendations.values()]\n})\nsubmission.to_csv('submission.csv', index=False)\nprint(f\"✓ Saved vedant_submission.csv\")\nprint(submission.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T19:17:56.095457Z","iopub.execute_input":"2025-11-03T19:17:56.095713Z","iopub.status.idle":"2025-11-03T19:26:43.692785Z","shell.execute_reply.started":"2025-11-03T19:17:56.095694Z","shell.execute_reply":"2025-11-03T19:26:43.691962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'customer_id': list(recommendations.keys()),\n    'prediction': [' '.join([str(int(x)).zfill(10) for x in v]) for v in recommendations.values()]\n})\nsubmission.to_csv('submission.csv', index=False)\nprint(f\"✓ Saved submission.csv\")\nprint(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T19:30:37.768298Z","iopub.execute_input":"2025-11-03T19:30:37.768905Z","iopub.status.idle":"2025-11-03T19:30:50.473360Z","shell.execute_reply.started":"2025-11-03T19:30:37.768882Z","shell.execute_reply":"2025-11-03T19:30:50.472596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}