{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ================================\n# 📦 1. Import Libraries\n# ================================","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom tqdm.notebook import tqdm\n\nimport warnings\nwarnings.simplefilter('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-01T04:55:33.760744Z","iopub.execute_input":"2025-06-01T04:55:33.761223Z","iopub.status.idle":"2025-06-01T04:55:33.767042Z","shell.execute_reply.started":"2025-06-01T04:55:33.761191Z","shell.execute_reply":"2025-06-01T04:55:33.766082Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n# ================================\n# 📂 2. Load Data\n# ================================","metadata":{}},{"cell_type":"code","source":"articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\nsample_submission = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T04:55:34.601623Z","iopub.execute_input":"2025-06-01T04:55:34.601949Z","iopub.status.idle":"2025-06-01T04:56:56.186697Z","shell.execute_reply.started":"2025-06-01T04:55:34.601928Z","shell.execute_reply":"2025-06-01T04:56:56.185463Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ================================\n# 🔍 3. Dataset Overview\n# ================================","metadata":{}},{"cell_type":"code","source":"print(\"\\n--- 1.1. articles の概要 ---\")\nprint(articles.info())\nprint(\"\\n--- articles の最初の5行 ---\")\ndisplay(articles.head())\nprint(\"\\n--- articles の欠損値 ---\")\nprint(articles.isnull().sum())\nprint(\"\\n--- articles のユニークなカテゴリカルな値 ---\")\nfor col in ['product_type_name', 'product_group_name', 'graphical_appearance_name',\n            'colour_group_name', 'perceived_colour_value_name', 'perceived_colour_master_name',\n            'department_name', 'index_name', 'index_group_name', 'section_name', 'garment_group_name']:\n    print(f\"{col}: {articles[col].nunique()} ユニークな値\")\n\nprint(\"\\n--- 1.2. customers の概要 ---\")\nprint(customers.info())\nprint(\"\\n--- customers の最初の5行 ---\")\ndisplay(customers.head())\nprint(\"\\n--- customers の欠損値 ---\")\nprint(customers.isnull().sum())\nprint(\"\\n--- customers のユニークなカテゴリカルな値 ---\")\nfor col in ['FN', 'Active', 'club_member_status', 'fashion_news_frequency']:\n    print(f\"{col}: {customers[col].nunique()} ユニークな値\")\n\nprint(\"\\n--- 1.3. transactions の概要 ---\")\nprint(transactions.info())\nprint(\"\\n--- transactions の最初の5行 ---\")\ndisplay(transactions.head())\nprint(\"\\n--- transactions の欠損値 ---\")\nprint(transactions.isnull().sum())\nprint(\"\\n--- transactions の期間 ---\")\nprint(f\"最小取引日: {transactions['t_dat'].min()}\")\nprint(f\"最大取引日: {transactions['t_dat'].max()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T04:48:33.733964Z","iopub.execute_input":"2025-06-01T04:48:33.734397Z","iopub.status.idle":"2025-06-01T04:48:43.189690Z","shell.execute_reply.started":"2025-06-01T04:48:33.734357Z","shell.execute_reply":"2025-06-01T04:48:43.188494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"\\n--- 2.1. customers の欠損値 ('FN', 'Active', 'age', 'fashion_news_frequency') の分析 ---\")\n# 'FN'と'Active'の欠損値はNaNであり、0.0として扱えるか検討\n# customers_df['FN'] = customers_df['FN'].fillna(0).astype(int)\n# customers_df['Active'] = customers_df['Active'].fillna(0).astype(int)\n# 'fashion_news_frequency'の欠損値は不明として扱うか、最頻値などで補完するか検討\n# 'age'の欠損値は少ないので、平均値や中央値で補完するか、欠損ユーザーを削除するか検討\n\nprint(\"\\n--- 2.2. articles の詳細説明 (detail_desc) の欠損値 ---\")\n# detail_descの欠損値は商品によって異なる可能性があるため、テキスト分析時に考慮\nprint(f\"detail_desc の欠損値数: {articles['detail_desc'].isnull().sum()}\")\n\nprint(\"\\n--- 2.3. 重複データの確認 ---\")\nprint(f\"transactions の重複行数: {transactions.duplicated().sum()}\")\n# 重複行が存在する場合、それが意味のある重複（例：同じ日に同じ商品を複数購入）なのか、\n# データ入力エラーなのかを確認し、必要に応じて削除または集約します。","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T04:29:03.287113Z","iopub.execute_input":"2025-06-01T04:29:03.287911Z","iopub.status.idle":"2025-06-01T04:29:39.953421Z","shell.execute_reply.started":"2025-06-01T04:29:03.287885Z","shell.execute_reply":"2025-06-01T04:29:39.952460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n--- 3.1. ユーザーごとのインタラクション数 ---\")\nuser_interactions = transactions.groupby('customer_id').size().reset_index(name='interaction_count')\n\nplt.figure(figsize=(10, 6))\nsns.histplot(user_interactions['interaction_count'], bins=50, log_scale=True)\nplt.title('Distribution of Interactions per User (Log Scale)') #ユーザーごとのインタラクション数分布 (Log Scale)\nplt.xlabel('Number of Interactions') #インタラクション数\nplt.ylabel('Number of Users') #ユーザー数\nplt.grid(True, which=\"both\", ls=\"--\", c=\"0.7\")\nplt.show()\n\nprint(f\"平均インタラクション数: {user_interactions['interaction_count'].mean():.2f}\")\nprint(f\"中央値インタラクション数: {user_interactions['interaction_count'].median():.2f}\")\nprint(f\"インタラクション数1のユーザー数: {user_interactions[user_interactions['interaction_count'] == 1].shape[0]}\")\nprint(f\"インタラクション数1のユーザー割合: {user_interactions[user_interactions['interaction_count'] == 1].shape[0] / user_interactions.shape[0] * 100:.2f}%\")\n\nprint(\"\\n--- 3.2. ユーザーの年齢分布 ---\")\nplt.figure(figsize=(10, 6))\nsns.histplot(customers['age'].dropna(), bins=30, kde=True)\nplt.title('Customer Age Distribution') #顧客の年齢分布\nplt.xlabel('Age') #年齢\nplt.ylabel('Number of Customers') #顧客数\nplt.show()\n\nprint(\"\\n--- 3.3. クラブメンバーシップのステータスとファッションニュース受信頻度 ---\")\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nsns.countplot(data=customers, x='club_member_status')\nplt.title('Club Member Status') #クラブメンバーのステータス\nplt.subplot(1, 2, 2)\nsns.countplot(data=customers, x='fashion_news_frequency')\nplt.title('Fashion News Frequency') #ファッションニュースの受信頻度\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T04:17:41.924599Z","iopub.execute_input":"2025-06-01T04:17:41.924886Z","iopub.status.idle":"2025-06-01T04:18:03.790040Z","shell.execute_reply.started":"2025-06-01T04:17:41.924866Z","shell.execute_reply":"2025-06-01T04:18:03.788992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n--- 4.1. アイテムごとのインタラクション数 ---\")\nitem_interactions = transactions.groupby('article_id').size().reset_index(name='interaction_count')\n\nplt.figure(figsize=(10, 6))\nsns.histplot(item_interactions['interaction_count'], bins=50, log_scale=True)\nplt.title('Distribution of Interactions per Item (Log Scale)') #アイテムごとのインタラクション数分布 (Log Scale)\nplt.xlabel('Number of Interactions') #インタラクション数\nplt.ylabel('Number of Items') #アイテム数\nplt.grid(True, which=\"both\", ls=\"--\", c=\"0.7\")\nplt.show()\n\nprint(f\"平均インタラクション数: {item_interactions['interaction_count'].mean():.2f}\")\nprint(f\"中央値インタラクション数: {item_interactions['interaction_count'].median():.2f}\")\nprint(f\"インタラクション数1のアイテム数: {item_interactions[item_interactions['interaction_count'] == 1].shape[0]}\")\nprint(f\"インタラクション数1のアイテム割合: {item_interactions[item_interactions['interaction_count'] == 1].shape[0] / item_interactions.shape[0] * 100:.2f}%\")\n\nprint(\"\\n--- 4.2. 人気のある商品タイプ/グループ ---\")\n# articles_df と transactions_df を結合して分析\nmerged_transactions_articles = pd.merge(transactions, articles, on='article_id', how='left')\n\nplt.figure(figsize=(12, 6))\ntop_product_types = merged_transactions_articles['product_type_name'].value_counts().head(10)\nsns.barplot(x=top_product_types.index, y=top_product_types.values)\nplt.title('Top 10 Product Types by Purchase Count') #上位10商品タイプ別の購入数\nplt.xlabel('Product Type') #商品タイプ\nplt.ylabel('Purchase Count') #購入数\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()\n\nplt.figure(figsize=(12, 6))\ntop_product_groups = merged_transactions_articles['product_group_name'].value_counts().head(10)\nsns.barplot(x=top_product_groups.index, y=top_product_groups.values)\nplt.title('Top 10 Product Groups by Purchase Count') #上位10商品グループ別の購入数\nplt.xlabel('Product Group') #商品グループ\nplt.ylabel('Purchase Count') #購入数\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()\n\nprint(\"\\n--- 4.3. 色グループの人気度 ---\")\nplt.figure(figsize=(12, 6))\ntop_colors = merged_transactions_articles['colour_group_name'].value_counts().head(10)\nsns.barplot(x=top_colors.index, y=top_colors.values)\nplt.title('Top 10 Color Groups by Purchase Count') #上位10色グループ別の購入数\nplt.xlabel('Color Group') #色グループ\nplt.ylabel('Purchase Count') #購入数\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T04:19:51.331574Z","iopub.execute_input":"2025-06-01T04:19:51.331928Z","iopub.status.idle":"2025-06-01T04:20:25.624476Z","shell.execute_reply.started":"2025-06-01T04:19:51.331903Z","shell.execute_reply":"2025-06-01T04:20:25.623479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# t_datをdatetime型に変換\ntransactions['t_dat'] = pd.to_datetime(transactions['t_dat'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:24:10.813361Z","iopub.execute_input":"2025-06-01T05:24:10.813784Z","iopub.status.idle":"2025-06-01T05:24:14.432569Z","shell.execute_reply.started":"2025-06-01T05:24:10.813756Z","shell.execute_reply":"2025-06-01T05:24:14.431466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n--- 5.1. 時間経過に伴うインタラクション数 ---\")\ntransactions['transaction_date'] = transactions['t_dat'].dt.date\ndaily_transactions = transactions.groupby('transaction_date').size()\n\nplt.figure(figsize=(15, 7))\ndaily_transactions.plot()\nplt.title('Daily Transaction Count') #日ごとの取引数\nplt.xlabel('Date') #日付\nplt.ylabel('Transaction Count') #取引数\nplt.show()\n\nprint(\"\\n--- 5.2. 価格分布 ---\")\nplt.figure(figsize=(10, 6))\nsns.histplot(transactions['price'], bins=50)\nplt.title('Distribution of Item Prices') #商品の価格分布\nplt.xlabel('Price') #価格\nplt.ylabel('Number of Transactions') #取引数\nplt.show()\n\nprint(\"\\n--- 5.3. 販売チャネル別の取引数 ---\")\nplt.figure(figsize=(7, 5))\nsns.countplot(data=transactions, x='sales_channel_id')\nplt.title('Transaction Count by Sales Channel') #販売チャネル別の取引数\nplt.xlabel('Sales Channel ID') #販売チャネルID\nplt.ylabel('Transaction Count') #取引数\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T04:23:02.969625Z","iopub.execute_input":"2025-06-01T04:23:02.969980Z","iopub.status.idle":"2025-06-01T04:23:38.597104Z","shell.execute_reply.started":"2025-06-01T04:23:02.969957Z","shell.execute_reply":"2025-06-01T04:23:38.596086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n--- 6.1. ユーザーあたりのユニークアイテム購入数 ---\")\nuser_unique_items = transactions.groupby('customer_id')['article_id'].nunique().reset_index(name='unique_item_count')\n\nplt.figure(figsize=(10, 6))\nsns.histplot(user_unique_items['unique_item_count'], bins=50, log_scale=True)\nplt.title('Distribution of Unique Items Purchased per User (Log Scale)') #ユーザーごとのユニークアイテム購入数分布 (Log Scale)\nplt.xlabel('Number of Unique Items') #ユニークアイテム数\nplt.ylabel('Number of Users') #ユーザー数\nplt.grid(True, which=\"both\", ls=\"--\", c=\"0.7\")\nplt.show()\n\nprint(f\"平均ユニークアイテム購入数: {user_unique_items['unique_item_count'].mean():.2f}\")\nprint(f\"中央値ユニークアイテム購入数: {user_unique_items['unique_item_count'].median():.2f}\")\nprint(f\"ユニークアイテム購入数1のユーザー数: {user_unique_items[user_unique_items['unique_item_count'] == 1].shape[0]}\")\nprint(f\"ユニークアイテム購入数1のユーザー割合: {user_unique_items[user_unique_items['unique_item_count'] == 1].shape[0] / user_unique_items.shape[0] * 100:.2f}%\")\n\nprint(\"\\n--- 6.2. 顧客が最も購入した商品の種類 ---\")\n# 各顧客が最も多く購入したproduct_group_nameを特定\ncustomer_top_product_group = merged_transactions_articles.groupby('customer_id')['product_group_name'].agg(lambda x: x.mode()[0] if not x.mode().empty else np.nan)\nplt.figure(figsize=(12, 6))\ncustomer_top_product_group.value_counts().head(10).plot(kind='bar')\nplt.title('Distribution of Most Purchased Product Groups by Customer (Top 10)') #顧客が最も購入した商品グループの分布 (上位10)\nplt.xlabel('Product Group') #商品グループ\nplt.ylabel('Number of Customers') #顧客数\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()\n\nprint(\"\\n--- 6.3. 最新のトランザクション日の確認と、評価期間の設定 ---\")\nlatest_transaction_date = transactions['t_dat'].max()\nprint(f\"Latest transaction date in the dataset: {latest_transaction_date}\") #データセットの最新取引日\n# 例として、最後のN週間/日をテストセットとして使用する場合の考慮\n# 例えば、過去2週間のデータで学習し、その後の1週間の購買を予測するなど。\n# モデルの評価時に、推薦リストの12個にユーザーの実際の購買行動をどれだけ含められるかを確認します。","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T04:24:30.683775Z","iopub.execute_input":"2025-06-01T04:24:30.684134Z","iopub.status.idle":"2025-06-01T04:29:03.285608Z","shell.execute_reply.started":"2025-06-01T04:24:30.684109Z","shell.execute_reply":"2025-06-01T04:29:03.284480Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ================================\n# 🔍 4. More Explore\n# ================================","metadata":{}},{"cell_type":"code","source":"# customer_idとarticle_idを結合\n# merged_transactions_articles = pd.merge(transactions, articles, on='article_id', how='left')\n# merged_transactions_full = pd.merge(merged_transactions_articles, customers, on='customer_id', how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T04:56:56.188493Z","iopub.execute_input":"2025-06-01T04:56:56.188766Z","iopub.status.idle":"2025-06-01T04:59:49.618399Z","shell.execute_reply.started":"2025-06-01T04:56:56.188744Z","shell.execute_reply":"2025-06-01T04:59:49.616521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- ユーザー側の観点 ---\nprint(\"\\n--- ユーザー側の観点からのEDA ---\")\n\n# 1. 購入頻度（週ごと、月ごと）\nprint(\"\\n--- 1.1. ユーザーごとの購入頻度（週ごと、月ごと） ---\")\n# ユーザーごとの最初の購入日と最後の購入日を計算\nuser_activity = transactions.groupby('customer_id')['t_dat'].agg(['min', 'max']).reset_index()\nuser_activity['duration_days'] = (user_activity['max'] - user_activity['min']).dt.days\n\n# 総インタラクション数\nuser_total_interactions = transactions.groupby('customer_id').size().reset_index(name='total_interactions')\nuser_activity = pd.merge(user_activity, user_total_interactions, on='customer_id', how='left')\n\n# 週ごと、月ごとの平均購入頻度を概算 (アクティブ期間が0のユーザーを除く)\nuser_activity['avg_weekly_purchase'] = user_activity.apply(\n    lambda row: (row['total_interactions'] / (row['duration_days'] / 7)) if row['duration_days'] > 0 else row['total_interactions'], axis=1\n)\nuser_activity['avg_monthly_purchase'] = user_activity.apply(\n    lambda row: (row['total_interactions'] / (row['duration_days'] / 30.4375)) if row['duration_days'] > 0 else row['total_interactions'], axis=1\n)\n\nplt.figure(figsize=(15, 6))\nplt.subplot(1, 2, 1)\nsns.histplot(user_activity['avg_weekly_purchase'].dropna(), bins=50, log_scale=True)\nplt.title('Average Weekly Purchase Frequency per User') #ユーザーごとの週平均購入頻度\nplt.xlabel('Average Weekly Purchases') #週平均購入数\nplt.ylabel('Number of Users') #ユーザー数\n\nplt.subplot(1, 2, 2)\nsns.histplot(user_activity['avg_monthly_purchase'].dropna(), bins=50, log_scale=True)\nplt.title('Average Monthly Purchase Frequency per User') #ユーザーごとの月平均購入頻度\nplt.xlabel('Average Monthly Purchases') #月平均購入数\nplt.ylabel('Number of Users') #ユーザー数\nplt.tight_layout()\nplt.show()\n\nprint(\"週平均購入頻度の要約統計量:\\n\", user_activity['avg_weekly_purchase'].describe())\nprint(\"月平均購入頻度の要約統計量:\\n\", user_activity['avg_monthly_purchase'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:36:06.919634Z","iopub.execute_input":"2025-06-01T05:36:06.920103Z","iopub.status.idle":"2025-06-01T05:37:12.330265Z","shell.execute_reply.started":"2025-06-01T05:36:06.920071Z","shell.execute_reply":"2025-06-01T05:37:12.328763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. 最後の購入からの経過日数\nprint(\"\\n--- 2. 最後の購入からの経過日数 ---\")\n# データセットの最新日を取得\nlatest_data_date = transactions['t_dat'].max()\nuser_last_purchase = transactions.groupby('customer_id')['t_dat'].max().reset_index()\nuser_last_purchase['days_since_last_purchase'] = (latest_data_date - user_last_purchase['t_dat']).dt.days\n\nplt.figure(figsize=(10, 6))\nsns.histplot(user_last_purchase['days_since_last_purchase'], bins=50, kde=True)\nplt.title('Days Since Last Purchase per User') #ユーザーごとの最後の購入からの経過日数\nplt.xlabel('Days Elapsed') #経過日数\nplt.ylabel('Number of Users') #ユーザー数\nplt.show()\n\nprint(\"最後の購入からの経過日数の要約統計量:\\n\", user_last_purchase['days_since_last_purchase'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:37:12.332123Z","iopub.execute_input":"2025-06-01T05:37:12.332506Z","iopub.status.idle":"2025-06-01T05:37:31.827035Z","shell.execute_reply.started":"2025-06-01T05:37:12.332471Z","shell.execute_reply":"2025-06-01T05:37:31.826085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. 平均購入価格、購入合計金額\nprint(\"\\n--- 3. 平均購入価格、購入合計金額 ---\")\nuser_purchase_stats = transactions.groupby('customer_id')['price'].agg(['mean', 'sum']).reset_index()\nuser_purchase_stats.rename(columns={'mean': 'avg_purchase_price', 'sum': 'total_purchase_amount'}, inplace=True)\n\nplt.figure(figsize=(15, 6))\nplt.subplot(1, 2, 1)\nsns.histplot(user_purchase_stats['avg_purchase_price'], bins=50, kde=True)\nplt.title('Average Purchase Price per User') #ユーザーごとの平均購入価格\nplt.xlabel('Average Purchase Price') #平均購入価格\nplt.ylabel('Number of Users') #ユーザー数\n\nplt.subplot(1, 2, 2)\nsns.histplot(user_purchase_stats['total_purchase_amount'], bins=50, log_scale=True)\nplt.title('Total Purchase Amount per User (Log Scale)') #ユーザーごとの購入合計金額 (Log Scale)\nplt.xlabel('Total Purchase Amount') #購入合計金額\nplt.ylabel('Number of Users') #ユーザー数\nplt.tight_layout()\nplt.show()\n\nprint(\"ユーザーごとの平均購入価格の要約統計量:\\n\", user_purchase_stats['avg_purchase_price'].describe())\nprint(\"ユーザーごとの購入合計金額の要約統計量:\\n\", user_purchase_stats['total_purchase_amount'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:15:52.414298Z","iopub.execute_input":"2025-06-01T05:15:52.414722Z","iopub.status.idle":"2025-06-01T05:16:14.104471Z","shell.execute_reply.started":"2025-06-01T05:15:52.414694Z","shell.execute_reply":"2025-06-01T05:16:14.103341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. 購入したアイテムのカテゴリ、色、グループの多様性\nprint(\"\\n--- 4. 購入したアイテムのカテゴリ、色、グループの多様性 ---\")\n# product_group_nameの多様性\nuser_product_group_diversity = merged_transactions_full.groupby('customer_id')['product_group_name'].nunique().reset_index(name='unique_product_groups')\nplt.figure(figsize=(10, 6))\nsns.histplot(user_product_group_diversity['unique_product_groups'], bins=range(1, user_product_group_diversity['unique_product_groups'].max() + 2), kde=False)\nplt.title('Diversity of Purchased Product Groups per User') #ユーザーごとの購入商品グループの多様性\nplt.xlabel('Number of Unique Product Groups') #ユニークな商品グループ数\nplt.ylabel('Number of Users') #ユーザー数\nplt.xticks(rotation=45, ha='right')\nplt.show()\nprint(\"ユニークな商品グループ数の要約統計量:\\n\", user_product_group_diversity['unique_product_groups'].describe())\n\n# colour_group_nameの多様性\nuser_color_diversity = merged_transactions_full.groupby('customer_id')['colour_group_name'].nunique().reset_index(name='unique_color_groups')\nplt.figure(figsize=(10, 6))\nsns.histplot(user_color_diversity['unique_color_groups'], bins=range(1, user_color_diversity['unique_color_groups'].max() + 2), kde=False)\nplt.title('Diversity of Purchased Color Groups per User') #ユーザーごとの購入色グループの多様性\nplt.xlabel('Number of Unique Color Groups') #ユニークな色グループ数\nplt.ylabel('Number of Users') #ユーザー数\nplt.xticks(rotation=45, ha='right')\nplt.show()\nprint(\"ユニークな色グループ数の要約統計量:\\n\", user_color_diversity['unique_color_groups'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:16:54.559778Z","iopub.execute_input":"2025-06-01T05:16:54.560176Z","iopub.status.idle":"2025-06-01T05:17:32.272698Z","shell.execute_reply.started":"2025-06-01T05:16:54.560147Z","shell.execute_reply":"2025-06-01T05:17:32.271364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 5. 年齢、会員ステータス、ファッションニュース購読状況\nprint(\"\\n--- 5. 年齢、会員ステータス、ファッションニュース購読状況 ---\")\n# 年齢分布はすでに確認済みだが、ここで再確認\nplt.figure(figsize=(15, 5))\nplt.subplot(1, 3, 1)\nsns.histplot(customers['age'].dropna(), bins=30, kde=True)\nplt.title('Customer Age Distribution') #顧客の年齢分布\nplt.xlabel('Age') #年齢\nplt.ylabel('Number of Customers') #顧客数\n\nplt.subplot(1, 3, 2)\nsns.countplot(data=customers, x='club_member_status', palette='viridis')\nplt.title('Club Member Status') #クラブメンバーのステータス\nplt.xlabel('Status') #ステータス\nplt.ylabel('Number of Customers') #顧客数\n\nplt.subplot(1, 3, 3)\nsns.countplot(data=customers, x='fashion_news_frequency', palette='magma')\nplt.title('Fashion News Subscription Frequency') #ファッションニュースの受信頻度\nplt.xlabel('Frequency') #受信頻度\nplt.ylabel('Number of Customers') #顧客数\nplt.tight_layout()\nplt.show()\n\n# FN, Active の値の分布\nprint(\"\\nFN (ファッションニュースレター購読) の分布:\\n\", customers['FN'].value_counts(dropna=False))\nprint(\"\\nActive (コミュニケーションアクティブ) の分布:\\n\", customers['Active'].value_counts(dropna=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:18:42.491861Z","iopub.execute_input":"2025-06-01T05:18:42.492369Z","iopub.status.idle":"2025-06-01T05:18:50.698955Z","shell.execute_reply.started":"2025-06-01T05:18:42.492256Z","shell.execute_reply":"2025-06-01T05:18:50.697776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- アイテム側の観点 ---\nprint(\"\\n--- アイテム側の観点からのEDA ---\")\n\n# 1. 人気度（購入回数、ユニークユーザー数）\nprint(\"\\n--- 1. 人気度（購入回数、ユニークユーザー数） ---\")\nitem_popularity = transactions.groupby('article_id').agg(\n    purchase_count=('customer_id', 'size'),\n    unique_user_count=('customer_id', 'nunique')\n).reset_index()\n\nplt.figure(figsize=(15, 6))\nplt.subplot(1, 2, 1)\nsns.histplot(item_popularity['purchase_count'], bins=50, log_scale=True)\nplt.title('Number of Purchases per Item (Log Scale)') #アイテムごとの購入回数 (Log Scale)\nplt.xlabel('Purchase Count') #購入回数\nplt.ylabel('Number of Items') #アイテム数\n\nplt.subplot(1, 2, 2)\nsns.histplot(item_popularity['unique_user_count'], bins=50, log_scale=True)\nplt.title('Number of Unique Users per Item (Log Scale)') #アイテムごとのユニークユーザー数 (Log Scale)\nplt.xlabel('Unique User Count') #ユニークユーザー数\nplt.ylabel('Number of Items') #アイテム数\nplt.tight_layout()\nplt.show()\n\nprint(\"アイテムごとの購入回数の要約統計量:\\n\", item_popularity['purchase_count'].describe())\nprint(\"アイテムごとのユニークユーザー数の要約統計量:\\n\", item_popularity['unique_user_count'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:19:59.707902Z","iopub.execute_input":"2025-06-01T05:19:59.709048Z","iopub.status.idle":"2025-06-01T05:20:22.086780Z","shell.execute_reply.started":"2025-06-01T05:19:59.708984Z","shell.execute_reply":"2025-06-01T05:20:22.085656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. 発売日からの経過日数 (articles.csvに発売日情報がないため、ここでは割愛)\n# もし発売日情報が別途あれば、以下の計算を行う\n# print(\"\\n--- 2. 発売日からの経過日数 ---\")\n# # 仮に'release_date'カラムがあるとして\n# articles_df['release_date'] = pd.to_datetime(articles_df['release_date'])\n# current_date = pd.to_datetime('2024-01-01') # またはtransactions_dfの最新日など\n# articles_df['days_since_release'] = (current_date - articles_df['release_date']).dt.days\n# plt.figure(figsize=(10, 6))\n# sns.histplot(articles_df['days_since_release'].dropna(), bins=50, kde=True)\n# plt.title('アイテムの発売日からの経過日数')\n# plt.xlabel('経過日数')\n# plt.ylabel('アイテム数')\n# plt.show()\n# print(\"発売日からの経過日数の要約統計量:\\n\", articles_df['days_since_release'].describe())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. 平均価格 (商品の価格はtransactions_dfにあるため、ここで再確認)\nprint(\"\\n--- 3. 平均価格 ---\")\n# article_idごとの平均価格を計算 (複数のトランザクションで価格が異なる可能性を考慮)\nitem_avg_price = transactions.groupby('article_id')['price'].mean().reset_index(name='avg_item_price')\nplt.figure(figsize=(10, 6))\nsns.histplot(item_avg_price['avg_item_price'], bins=50, kde=True)\nplt.title('Average Price per Item') #アイテムごとの平均価格\nplt.xlabel('Average Price') #平均価格\nplt.ylabel('Number of Items') #アイテム数\nplt.show()\nprint(\"アイテムごとの平均価格の要約統計量:\\n\", item_avg_price['avg_item_price'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:20:44.961986Z","iopub.execute_input":"2025-06-01T05:20:44.962362Z","iopub.status.idle":"2025-06-01T05:20:46.970390Z","shell.execute_reply.started":"2025-06-01T05:20:44.962338Z","shell.execute_reply":"2025-06-01T05:20:46.969596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. カテゴリ、色、グループ、詳細説明からのテキスト特徴量（TF-IDF, Word2Vecなど）\nprint(\"\\n--- 4. カテゴリ、色、グループ、詳細説明からのテキスト特徴量 ---\")\n# カテゴリ、色、グループの分布はすでに確認済みだが、ここで改めてユニーク数を確認\nprint(articles[['product_type_name', 'product_group_name', 'colour_group_name', 'detail_desc']].head())\nprint(\"\\nproduct_type_name のユニーク数:\", articles['product_type_name'].nunique())\nprint(\"product_group_name のユニーク数:\", articles['product_group_name'].nunique())\nprint(\"colour_group_name のユニーク数:\", articles['colour_group_name'].nunique())\n\n# detail_desc の欠損値確認（テキスト分析の前に処理が必要）\nprint(f\"detail_desc の欠損値数: {articles['detail_desc'].isnull().sum()}\")\n\n# detail_desc のワードクラウドや頻出単語分析は、別途NLPライブラリ (NLTK, spaCy, scikit-learn) を用いて行う\n# 例: 最初のいくつかの詳細説明を表示\nprint(\"\\n最初の5つの詳細説明の例:\\n\")\nfor i, desc in enumerate(articles['detail_desc'].head()):\n    print(f\"Article {articles['article_id'].iloc[i]}: {desc}\")\n\n# 簡単な単語頻度分析（例：最も一般的な単語）\nfrom collections import Counter\nimport re\n\n# 小文字に変換し、数字と句読点を削除\narticles['cleaned_desc'] = articles['detail_desc'].fillna('').astype(str).apply(lambda x: re.sub(r'[^a-zA-Z\\s]', '', x).lower())\nall_words = ' '.join(articles['cleaned_desc']).split()\nword_counts = Counter(all_words)\nprint(\"\\n詳細説明で最も頻繁に現れる単語 (Top 20):\\n\", word_counts.most_common(20))\n# ここで'a', 'the', 'is'などのストップワードが見られる。実際の分析ではこれらを除外する。","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:21:18.699696Z","iopub.execute_input":"2025-06-01T05:21:18.700065Z","iopub.status.idle":"2025-06-01T05:21:19.728314Z","shell.execute_reply.started":"2025-06-01T05:21:18.700038Z","shell.execute_reply":"2025-06-01T05:21:19.727291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- インタラクション側の観点 ---\nprint(\"\\n--- インタラクション側の観点からのEDA ---\")\n\n# 1. 購入回数（リピート購入）\nprint(\"\\n--- 1. 購入回数（リピート購入） ---\")\n# ユーザーとアイテムの組み合わせごとの購入回数\nrepeat_purchases = transactions.groupby(['customer_id', 'article_id']).size().reset_index(name='purchase_count')\nrepeat_purchases_gt_1 = repeat_purchases[repeat_purchases['purchase_count'] > 1]\n\nprint(f\"リピート購入された (同じユーザーが同じ商品を複数回購入) ユニークな組み合わせの数: {repeat_purchases_gt_1.shape[0]}\")\nprint(f\"総インタラクションに対するリピート購入の割合: {repeat_purchases_gt_1['purchase_count'].sum() / transactions.shape[0] * 100:.2f}%\")\n\nplt.figure(figsize=(14, 8))\nsns.histplot(repeat_purchases_gt_1['purchase_count'], bins=range(2, repeat_purchases['purchase_count'].max() + 2), discrete=True)\nplt.title('Repeat Purchase Count per User-Item Combination') #ユーザーとアイテムの組み合わせごとのリピート購入回数\nplt.xlabel('Purchase Count') #購入回数\nplt.ylabel('Number of Combinations') #組み合わせ数\n\n# X軸の目盛りを調整して、重複を減らす\nmax_purchase_count = repeat_purchases['purchase_count'].max()\nif max_purchase_count < 10: # 例: 最大9回までなら全て表示\n    tick_interval = 1\nelif max_purchase_count < 25: # 例: 最大24回までなら2回おきに表示\n    tick_interval = 2\nelif max_purchase_count < 50: # 例: 最大49回までなら5回おきに表示\n    tick_interval = 5\nelse: # それ以上なら10回おきに表示\n    tick_interval = 10\n\nplt.xticks(np.arange(2, max_purchase_count + 1, tick_interval), rotation=45, ha='right') # 目盛り間隔を調整し、45度回転させる\n\nplt.grid(axis='y', linestyle='--', alpha=0.7) # グリッド線を追加して読みやすくする\nplt.tight_layout() # レイアウトを自動調整","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:29:06.657123Z","iopub.execute_input":"2025-06-01T05:29:06.657537Z","iopub.status.idle":"2025-06-01T05:29:42.892031Z","shell.execute_reply.started":"2025-06-01T05:29:06.657510Z","shell.execute_reply":"2025-06-01T05:29:42.890822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. 購入の曜日、時間帯\nprint(\"\\n--- 2. 購入の曜日、時間帯 ---\")\ntransactions['day_of_week'] = transactions['t_dat'].dt.day_name()\ntransactions['hour_of_day'] = transactions['t_dat'].dt.hour\n\nplt.figure(figsize=(15, 6))\nplt.subplot(1, 2, 1)\nsns.countplot(data=transactions, x='day_of_week', order=['Monday', 'Tuesday', 'Wednesday', 'Thursday', 'Friday', 'Saturday', 'Sunday'], palette='coolwarm')\nplt.title('Purchase Count by Day of Week') #曜日ごとの購入数\nplt.xlabel('Day of Week') #曜日\nplt.ylabel('Purchase Count') #購入数\n\nplt.subplot(1, 2, 2)\nsns.histplot(transactions['hour_of_day'], bins=24, kde=False)\nplt.title('Purchase Count by Hour of Day') #時間帯ごとの購入数\nplt.xlabel('Hour of Day (24-hour format)') #時間（24時間表記)\nplt.ylabel('Purchase Count') #購入数\nplt.xticks(range(0, 24))\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:24:23.295270Z","iopub.execute_input":"2025-06-01T05:24:23.295633Z","iopub.status.idle":"2025-06-01T05:25:06.406155Z","shell.execute_reply.started":"2025-06-01T05:24:23.295612Z","shell.execute_reply":"2025-06-01T05:25:06.405207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. 販売チャネル（店舗 vs. オンライン）\nprint(\"\\n--- 3. 販売チャネル（店舗 vs. オンライン） ---\")\nplt.figure(figsize=(7, 5))\nsns.countplot(data=transactions, x='sales_channel_id', palette='pastel')\nplt.title('Transaction Count by Sales Channel') #販売チャネル別の取引数\nplt.xlabel('Sales Channel ID (1=Store, 2=Online)') #販売チャネルID (1=店舗, 2=オンライン)\nplt.ylabel('Transaction Count') #取引数\nplt.show()\n\nprint(\"販売チャネル別の取引数:\\n\", transactions['sales_channel_id'].value_counts())\nprint(f\"オンライン購入の割合: {transactions['sales_channel_id'].value_counts(normalize=True).get(2, 0) * 100:.2f}%\")\nprint(f\"店舗購入の割合: {transactions['sales_channel_id'].value_counts(normalize=True).get(1, 0) * 100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T05:25:06.407666Z","iopub.execute_input":"2025-06-01T05:25:06.408046Z","iopub.status.idle":"2025-06-01T05:25:10.352207Z","shell.execute_reply.started":"2025-06-01T05:25:06.407993Z","shell.execute_reply":"2025-06-01T05:25:10.351300Z"}},"outputs":[],"execution_count":null}]}