{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import dask.dataframe as dd\n\n# Load the datasets\ntransactions = dd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ncustomers = dd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\narticles = dd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\n\n# Display the first few rows to confirm the datasets are loaded correctly\nprint(\"Transactions Sample:\")\nprint(transactions.head())\n\nprint(\"Customers Sample:\")\nprint(customers.head())\n\nprint(\"Articles Sample:\")\nprint(articles.head())\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-03T18:31:32.920197Z","iopub.execute_input":"2025-10-03T18:31:32.920549Z","iopub.status.idle":"2025-10-03T18:31:38.303332Z","shell.execute_reply.started":"2025-10-03T18:31:32.920512Z","shell.execute_reply":"2025-10-03T18:31:38.302469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge transactions with articles to get product details (prod_name, product_type)\ntransactions_with_products = transactions.merge(articles[['article_id', 'prod_name', 'product_type_name']], on='article_id', how='left')\n\n# Display the merged data to ensure correctness\nprint(\"Merged Transactions with Product Details:\")\nprint(transactions_with_products.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T18:32:10.772053Z","iopub.execute_input":"2025-10-03T18:32:10.772942Z","iopub.status.idle":"2025-10-03T18:32:12.734263Z","shell.execute_reply.started":"2025-10-03T18:32:10.772911Z","shell.execute_reply":"2025-10-03T18:32:12.733428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by customer_id and aggregate product names into lists (baskets)\ncustomer_baskets = transactions_with_products.groupby('customer_id')['prod_name'].apply(list)\n\n# Display the first few customer baskets\nprint(\"Customer Baskets (first few rows):\")\nprint(customer_baskets.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T18:32:32.611515Z","iopub.execute_input":"2025-10-03T18:32:32.612403Z","iopub.status.idle":"2025-10-03T18:33:54.481872Z","shell.execute_reply.started":"2025-10-03T18:32:32.612373Z","shell.execute_reply":"2025-10-03T18:33:54.480962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import dask.dataframe as dd\nfrom sklearn.preprocessing import OneHotEncoder\nimport pandas as pd\n\n# Load the datasets using Dask\ntransactions = dd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\narticles = dd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\n\n# Merge transactions with articles to get product details (prod_name)\ntransactions_with_products = transactions.merge(articles[['article_id', 'prod_name']], on='article_id', how='left')\n\n# Group by 'customer_id' and aggregate product names into lists (baskets)\ncustomer_baskets = transactions_with_products.groupby('customer_id')['prod_name'].apply(list).compute()\n\n# Convert the customer baskets to a list of products (flatten the lists)\nall_products = customer_baskets.apply(lambda x: [product for product in x]).explode()\n\n# Convert the flattened list of products to a Pandas DataFrame\nall_products_df = pd.DataFrame(all_products.tolist(), columns=[\"prod_name\"])\n\n# Initialize the OneHotEncoder\nencoder = OneHotEncoder(sparse=False)\n\n# Reshape the data to make sure it's a 2D array (OneHotEncoder needs this)\nencoded_baskets = encoder.fit_transform(all_products_df)\n\n# Convert the encoded baskets to a DataFrame for easy handling\nencoded_baskets_df = pd.DataFrame(encoded_baskets, columns=encoder.get_feature_names_out())\n\n# Display the one-hot encoded baskets\nprint(\"One-Hot Encoded Customer Baskets:\")\nprint(encoded_baskets_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T18:42:56.086809Z","iopub.execute_input":"2025-10-03T18:42:56.087114Z","iopub.status.idle":"2025-10-03T18:42:56.14736Z","shell.execute_reply.started":"2025-10-03T18:42:56.087095Z","shell.execute_reply":"2025-10-03T18:42:56.145957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}