{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"1. SATIŞ VE MÜŞTERİ ANALİZİ\n2. MÜŞTERİ SEGMENTASYONU (RFM Tekniği ile)\n3. SEPET ANALİZİ \n4. İŞBİRLİKÇİ FİLTRELEME\n    * 4.1 KULLANICI TABANLI ÖNERİ SİSTEMİ\n    * 4.2 ÜRÜN TABANLI ÖNERİ SİSTEMİ\n5. İÇERİK TABANLI FİLTRELEME\n    * 5.1 METİN TABANLI ÖNERİ SİSTEMİ\n    * 5.2 GÖRÜNTÜ TABANI ÖNERİ SİSTEMİ (CNN ile)","metadata":{}},{"cell_type":"markdown","source":"------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"# 1. SATIŞ VE MÜŞTERİ ANALİZİ","metadata":{}},{"cell_type":"code","source":"#  kütüpaneler \n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nimport os\nfrom collections import defaultdict\nfrom PIL import Image\n\npd.set_option('display.max_columns', None)\npd.set_option('display.width', 500)\npd.set_option('display.expand_frame_repr', False)\n\n\nimport warnings\nwarnings.simplefilter(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:42:00.545704Z","iopub.execute_input":"2023-09-04T16:42:00.546207Z","iopub.status.idle":"2023-09-04T16:42:00.554795Z","shell.execute_reply.started":"2023-09-04T16:42:00.546171Z","shell.execute_reply":"2023-09-04T16:42:00.553462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# veri seti indirme \n\ndf_trs = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\ndf_articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_cust = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:42:00.561867Z","iopub.execute_input":"2023-09-04T16:42:00.562351Z","iopub.status.idle":"2023-09-04T16:43:07.260935Z","shell.execute_reply.started":"2023-09-04T16:42:00.562316Z","shell.execute_reply":"2023-09-04T16:43:07.259215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Veri setlerini 'Kadin giyim' icerecek sekilde duzenleme\n\ndf_articles = df_articles[df_articles['index_group_name'] == 'Ladieswear']\ndf_trs = df_trs[df_trs['article_id'].isin(df_articles['article_id'])]\ndf_cust = df_cust[df_cust['customer_id'].isin(df_trs['customer_id'])]","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:43:07.264588Z","iopub.execute_input":"2023-09-04T16:43:07.264989Z","iopub.status.idle":"2023-09-04T16:43:19.506059Z","shell.execute_reply.started":"2023-09-04T16:43:07.264958Z","shell.execute_reply":"2023-09-04T16:43:19.504752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Toplam urun tipi - essiz urun kodu icin\n\ndf_articles['article_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:43:19.507821Z","iopub.execute_input":"2023-09-04T16:43:19.508223Z","iopub.status.idle":"2023-09-04T16:43:19.520888Z","shell.execute_reply.started":"2023-09-04T16:43:19.508188Z","shell.execute_reply":"2023-09-04T16:43:19.519484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dizindeki klasor ve dosyalari sayma\n\ntotal_folders = 0\ntotal_files = 0\n\nfolder_info = []\nimages_names = []\n\npath = \"../input/h-and-m-personalized-fashion-recommendations\"\n\nfor base, dirs, files in tqdm(os.walk(path)):\n    for directories in dirs:\n        folder_info.append((directories, \n                            len(os.listdir(os.path.join(base, directories)))))\n        total_folders = total_folders + 1\n    \n    for _files in files:\n        total_files = total_files + 1\n        if (len(_files.split(\".jpg\"))==2):\n            images_names.append(_files.split(\".jpg\")[0])\n","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:43:19.524081Z","iopub.execute_input":"2023-09-04T16:43:19.52496Z","iopub.status.idle":"2023-09-04T16:45:45.807377Z","shell.execute_reply.started":"2023-09-04T16:43:19.524916Z","shell.execute_reply":"2023-09-04T16:45:45.806001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 'image_names' listesini kullanarak veri seti olusturma\n\nimage_name_df = pd.DataFrame(images_names, columns = [\"image_name\"])\nimage_name_df[\"article_id\"] = image_name_df[\"image_name\"].apply(lambda x: int(x[1:]))\nimage_name_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:45:45.809377Z","iopub.execute_input":"2023-09-04T16:45:45.81001Z","iopub.status.idle":"2023-09-04T16:45:45.948774Z","shell.execute_reply.started":"2023-09-04T16:45:45.809974Z","shell.execute_reply":"2023-09-04T16:45:45.947244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_articles ve image_article_df birlestirme\n\nimage_article_df = df_articles[[\"article_id\", \n                                \"product_code\", \n                                \"product_group_name\", \n                                \"product_type_name\"]].merge(image_name_df, \n                                                            on=[\"article_id\"], \n                                                            how=\"left\")\nimage_article_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:45:45.950274Z","iopub.execute_input":"2023-09-04T16:45:45.95061Z","iopub.status.idle":"2023-09-04T16:45:46.031401Z","shell.execute_reply.started":"2023-09-04T16:45:45.95058Z","shell.execute_reply":"2023-09-04T16:45:46.030308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_article_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:45:46.032822Z","iopub.execute_input":"2023-09-04T16:45:46.033246Z","iopub.status.idle":"2023-09-04T16:45:46.04258Z","shell.execute_reply.started":"2023-09-04T16:45:46.033137Z","shell.execute_reply":"2023-09-04T16:45:46.040841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Kac urunun fotosu yok?\n\narticle_no_image_df = image_article_df.loc[image_article_df.image_name.isna()]\narticle_no_image_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:45:46.04453Z","iopub.execute_input":"2023-09-04T16:45:46.044896Z","iopub.status.idle":"2023-09-04T16:45:46.0696Z","shell.execute_reply.started":"2023-09-04T16:45:46.044867Z","shell.execute_reply":"2023-09-04T16:45:46.068078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # Urun gruplarının toplam satış miktarlarına göre pasta grafiği\n\nmerged_df = df_trs.merge(df_articles, on='article_id')\n\ngroup_sales = merged_df.groupby('product_group_name')['price'].sum()\n\ntotal_sales = group_sales.sum()\n\ngroup_percentages = (group_sales / total_sales) * 100\n\nthreshold = 3  # Define the threshold for grouping as percentage\nsmall_percentages = group_percentages[group_percentages < threshold]\ngroup_sales['Other'] = group_sales[small_percentages.index].sum()\ngroup_sales = group_sales.drop(small_percentages.index)\n\ncolor_palette = 'Set2'\ncolors = plt.get_cmap(color_palette)(range(len(group_sales)))\n\nplt.figure(figsize=(8, 8))\nplt.pie(group_sales, labels=group_sales.index, autopct='%1.1f%%', startangle=140, colors=colors, textprops={'fontsize': 12})  # Yazıları büyütmek için textprops kullanın\nplt.tight_layout()\n\nplt.gca().add_artist(plt.Circle((0, 0), 0.70, fc='white'))\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:45:46.071213Z","iopub.execute_input":"2023-09-04T16:45:46.071549Z","iopub.status.idle":"2023-09-04T16:46:16.134119Z","shell.execute_reply.started":"2023-09-04T16:45:46.07152Z","shell.execute_reply":"2023-09-04T16:46:16.132752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Farkli sezonlar icin satış miktarları\n\n\ndf_trs['t_dat'] = pd.to_datetime(df_trs['t_dat'])\n\nseasons = [\n    (pd.Timestamp('2018-09-20'), pd.Timestamp('2019-03-20')),\n    (pd.Timestamp('2019-03-21'), pd.Timestamp('2019-09-20')),\n    (pd.Timestamp('2019-09-21'), pd.Timestamp('2020-03-20')),\n    (pd.Timestamp('2020-03-21'), pd.Timestamp('2020-09-20'))\n]\n\nsales_by_season = []\nfor start_date, end_date in seasons:\n    sales = df_trs[(df_trs['t_dat'] >= start_date) & (df_trs['t_dat'] <= end_date)]['price'].sum()\n    sales_by_season.append(sales)\n\nseason_labels = ['Sep 2018 - Mar 2019', 'Mar 2019 - Sep 2019', 'Sep 2019 - Mar 2020', 'Mar 2020 - Sep 2020']\nfor i, label in enumerate(season_labels):\n    print(f\"{label}: {sales_by_season[i]:,.2f}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:46:16.138593Z","iopub.execute_input":"2023-09-04T16:46:16.139027Z","iopub.status.idle":"2023-09-04T16:46:23.125513Z","shell.execute_reply.started":"2023-09-04T16:46:16.138984Z","shell.execute_reply":"2023-09-04T16:46:23.123775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sezonda en çok satılan ürünler\n\n\ndf_trs['t_dat'] = pd.to_datetime(df_trs['t_dat'])\n\nmerged_data = df_trs.merge(df_articles[['article_id', 'prod_name']], on='article_id', how='left')\n\ndate_ranges = [\n    (pd.to_datetime('2018-09-20'), pd.to_datetime('2019-03-20')),\n    (pd.to_datetime('2019-03-21'), pd.to_datetime('2019-09-20')),\n    (pd.to_datetime('2019-09-21'), pd.to_datetime('2020-03-20')),\n    (pd.to_datetime('2020-03-21'), pd.to_datetime('2020-09-20'))\n]\n\nfig, axs = plt.subplots(2, 2, figsize=(15, 10))\naxs = axs.flatten()\n\nfor i, (start_date, end_date) in enumerate(date_ranges):\n    selected_data = merged_data[(merged_data['t_dat'] >= start_date) & (merged_data['t_dat'] <= end_date)]\n    \n    top_products_range = selected_data.groupby('prod_name')['price'].sum().nlargest(5).reset_index()\n    \n    sns.barplot(data=top_products_range, x='prod_name', y='price', ax=axs[i])\n    axs[i].set_xticklabels(axs[i].get_xticklabels(), rotation=45, ha='right')\n    axs[i].set_xlabel('Product Name')\n    axs[i].set_ylabel('Total Sales Amount')\n    axs[i].set_title(f'{start_date.strftime(\"%d-%b-%Y\")} - {end_date.strftime(\"%d-%b-%Y\")}')\n    axs[i].set_ylim(0, 2500)\n    axs[i].legend().set_visible(False)\n\nplt.tight_layout()\nplt.subplots_adjust(hspace=0.8)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:46:23.127477Z","iopub.execute_input":"2023-09-04T16:46:23.127834Z","iopub.status.idle":"2023-09-04T16:46:38.775591Z","shell.execute_reply.started":"2023-09-04T16:46:23.127803Z","shell.execute_reply":"2023-09-04T16:46:38.774218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Farklı renk ve ürün grupları kombinasyonlarına göre toplam satış miktarı\n\n\ndf_trs['t_dat'] = pd.to_datetime(df_trs['t_dat'])\n\nmerged_data = df_trs.merge(df_articles[['article_id', 'prod_name', 'colour_group_name', 'product_group_name']], on='article_id', how='left')\n\ncolor_group_sales = merged_data.groupby(['colour_group_name', 'product_group_name'])['price'].sum().reset_index()\n\nselected_product_groups = ['Garment Full body', 'Garment Upper body', 'Garment Lower body']\ncolor_group_sales_selected = color_group_sales[color_group_sales['product_group_name'].isin(selected_product_groups)]\n\ntop_colors = color_group_sales_selected.groupby('product_group_name', group_keys=False).apply(lambda x: x.nlargest(5, 'price')).reset_index(drop=True)\n\nsns.set_palette('Set2')\n\nfig, axs = plt.subplots(1, len(selected_product_groups), figsize=(18, 6))\nfor i, (product_group, data) in enumerate(top_colors.groupby('product_group_name')):\n    sizes = data['price']\n    labels = data['colour_group_name']\n    axs[i].pie(sizes, labels=labels, autopct=lambda p: '{:.1f}%'.format(p), startangle=140, textprops={'fontsize': 14})\n    axs[i].set_title(product_group, fontsize=16, fontweight='bold')\n\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:46:38.777604Z","iopub.execute_input":"2023-09-04T16:46:38.77838Z","iopub.status.idle":"2023-09-04T16:46:54.071024Z","shell.execute_reply.started":"2023-09-04T16:46:38.778335Z","shell.execute_reply":"2023-09-04T16:46:54.066701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sezona gore satis miktari \n\ndf_trs['t_dat'] = pd.to_datetime(df_trs['t_dat'])\n\nmonthly_sales_20_to_20 = df_trs[df_trs['t_dat'].dt.day == 20]\n\nmonthly_sales_grouped = monthly_sales_20_to_20.resample('M', on='t_dat')['price'].sum()\n\ndate_ranges = [\n    (pd.to_datetime('2018-09-20'), pd.to_datetime('2019-03-20')),\n    (pd.to_datetime('2019-03-21'), pd.to_datetime('2019-09-20')),\n    (pd.to_datetime('2019-09-21'), pd.to_datetime('2020-03-20')),\n    (pd.to_datetime('2020-03-21'), pd.to_datetime('2020-09-20'))\n]\n\nplt.figure(figsize=(10, 6))\n\nfor i, (start_date, end_date) in enumerate(date_ranges):\n    plt.axvspan(start_date, end_date, color='C{}'.format(i), alpha=0.2, label=f'{start_date.strftime(\"%b %d, %Y\")} - {end_date.strftime(\"%b %d, %Y\")}')\n    \nplt.plot(monthly_sales_grouped.index, monthly_sales_grouped, marker='o', color='blue', label='Sales Amount')\n\nplt.ylabel('Total Sales Amount')\nplt.xticks(rotation=45)\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:46:54.072987Z","iopub.execute_input":"2023-09-04T16:46:54.073376Z","iopub.status.idle":"2023-09-04T16:46:57.22328Z","shell.execute_reply.started":"2023-09-04T16:46:54.073342Z","shell.execute_reply":"2023-09-04T16:46:57.221376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Toplam satis miktari\n\ntotal_sales = df_trs['price'].sum()\ntotal_sales","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:46:57.225381Z","iopub.execute_input":"2023-09-04T16:46:57.225789Z","iopub.status.idle":"2023-09-04T16:46:57.293392Z","shell.execute_reply.started":"2023-09-04T16:46:57.225756Z","shell.execute_reply":"2023-09-04T16:46:57.291871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Toplam satilan urun sayisi\n\ntotal_quantity_sold = len(df_trs)\ntotal_quantity_sold","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:46:57.295078Z","iopub.execute_input":"2023-09-04T16:46:57.295704Z","iopub.status.idle":"2023-09-04T16:46:57.304307Z","shell.execute_reply.started":"2023-09-04T16:46:57.295668Z","shell.execute_reply":"2023-09-04T16:46:57.302808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Musteri yas gruplari pasta grafigi\n\nbins = [18, 30, 40, 50, 100]\nlabels = ['18-30', '31-40', '41-50', '51+']\n\ndf_cust['age_group'] = pd.cut(df_cust['age'], bins=bins, labels=labels, right=False)\n\ncustomers_by_age = df_cust['age_group'].value_counts()\n\nplt.figure(figsize=(8, 6))\ncolors = sns.color_palette('Set3', n_colors=len(customers_by_age))\nexplode = (0.1, 0, 0, 0)\n\nwedges, texts, autotexts = plt.pie(customers_by_age, labels=customers_by_age.index, colors=colors, autopct='%1.1f%%', startangle=140, explode=explode)\nplt.axis('equal') \n\nfor text in texts:\n    text.set_fontweight('bold')\n    text.set_fontsize(16)\n\ncentre_circle = plt.Circle((0,0),0.70,fc='white')\nfig = plt.gcf()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:46:57.305748Z","iopub.execute_input":"2023-09-04T16:46:57.306131Z","iopub.status.idle":"2023-09-04T16:46:57.620669Z","shell.execute_reply.started":"2023-09-04T16:46:57.306099Z","shell.execute_reply":"2023-09-04T16:46:57.61904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # Sezona gore en cok satilan urunler (urun gruplari karisik)\n\n\ndf_trs['t_dat'] = pd.to_datetime(df_trs['t_dat'])\n\nmerged_data = df_trs.merge(df_articles[['article_id', 'prod_name']], on='article_id', how='left')\n\ndate_ranges = [\n    (pd.to_datetime('2018-09-20'), pd.to_datetime('2019-03-20')),\n    (pd.to_datetime('2019-03-21'), pd.to_datetime('2019-09-20')),\n    (pd.to_datetime('2019-09-21'), pd.to_datetime('2020-03-20')),\n    (pd.to_datetime('2020-03-21'), pd.to_datetime('2020-09-20'))\n]\n\nsns.set_palette('Set2')\n\nfig, axs = plt.subplots(2, 2, figsize=(15, 10))\naxs = axs.flatten()\n\nfor i, (start_date, end_date) in enumerate(date_ranges):\n    selected_data = merged_data[(merged_data['t_dat'] >= start_date) & (merged_data['t_dat'] <= end_date)]\n    \n    top_products_range = selected_data.groupby('prod_name')['price'].sum().nlargest(5).reset_index()\n\n    sns.barplot(data=top_products_range, y='prod_name', x='price', ax=axs[i])\n    axs[i].set_xlabel('Total Sales Amount')\n    axs[i].set_title(f'{start_date.strftime(\"%d-%b-%Y\")} - {end_date.strftime(\"%d-%b-%Y\")}')\n    axs[i].set_xlim(0, 2500)\n    axs[i].set_ylabel('')\n    axs[i].legend().set_visible(False)\n\nplt.tight_layout()\nplt.subplots_adjust(hspace=0.4)\n\n# Show the plots\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:46:57.622968Z","iopub.execute_input":"2023-09-04T16:46:57.624393Z","iopub.status.idle":"2023-09-04T16:47:14.351371Z","shell.execute_reply.started":"2023-09-04T16:46:57.624321Z","shell.execute_reply":"2023-09-04T16:47:14.349965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Toplam müşteri sayısı\n\ntotal_customers = df_trs['customer_id'].nunique()\ntotal_customers","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:47:14.353223Z","iopub.execute_input":"2023-09-04T16:47:14.354669Z","iopub.status.idle":"2023-09-04T16:47:22.863364Z","shell.execute_reply.started":"2023-09-04T16:47:14.35462Z","shell.execute_reply":"2023-09-04T16:47:22.86186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Moda haberlerini duzenli veya aylik alanlar\n\nfashion_news_regularly_count = df_cust[df_cust['fashion_news_frequency'].isin(['Regularly', 'Monthly'])]['customer_id'].nunique()\nfashion_news_regularly_count","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:47:22.865301Z","iopub.execute_input":"2023-09-04T16:47:22.86567Z","iopub.status.idle":"2023-09-04T16:47:23.348758Z","shell.execute_reply.started":"2023-09-04T16:47:22.865639Z","shell.execute_reply":"2023-09-04T16:47:23.347387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# En cok harcama yapan ilk 10 musteri\n\n# Müşterileri müşteri numarasına göre gruplayıp toplam harcamayı ve ürün sayısını hesaplama\ncustomer_summary = df_trs.groupby('customer_id').agg({\n    'price': 'sum',\n    'article_id': 'count'\n}).reset_index()\n\n# Azalan sekilde siralama\ncustomer_summary = customer_summary.sort_values(by='price', ascending=False)\n\nprint(customer_summary[['customer_id', 'price', 'article_id']].head(10))","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:47:23.350522Z","iopub.execute_input":"2023-09-04T16:47:23.350889Z","iopub.status.idle":"2023-09-04T16:47:38.538352Z","shell.execute_reply.started":"2023-09-04T16:47:23.350857Z","shell.execute_reply":"2023-09-04T16:47:38.536925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # Ürün gruplarına göre ürünleri gruplayarak, her grup içindeki farklı ürünleri birleştirip tekilleştirme\n\ngrouped_products = defaultdict(list)\n\nfor _, row in image_article_df.iterrows():\n    grouped_products[row['product_group_name']].append(row['article_id'])\n\nunique_grouped_products = {group: set(ids) for group, ids in grouped_products.items()}","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:47:38.54028Z","iopub.execute_input":"2023-09-04T16:47:38.541229Z","iopub.status.idle":"2023-09-04T16:47:41.109216Z","shell.execute_reply.started":"2023-09-04T16:47:38.541175Z","shell.execute_reply":"2023-09-04T16:47:41.10804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# En cok satan urunleri bulma, urun grubuna gore\n\ndef plot_image_samples(image_article_df, product_group_name, cols=1, rows=-1):\n    image_path = \"../input/h-and-m-personalized-fashion-recommendations/images/\"\n    _df = image_article_df.loc[image_article_df.product_group_name==product_group_name]\n    article_ids = _df.article_id.values[0:cols*rows]\n    plt.figure(figsize=(2 + 3 * cols, 2 + 4 * rows))\n    for i in range(cols * rows):\n        article_id = (\"0\" + str(article_ids[i]))[-10:]\n        plt.subplot(rows, cols, i + 1)\n        plt.axis('off')\n        plt.title(f\"{product_group_name} {article_id[:3]}\\n{article_id}.jpg\")\n        image = Image.open(f\"{image_path}{article_id[:3]}/{article_id}.jpg\")\n        plt.imshow(image)\n\nplot_image_samples(image_article_df, \"Garment Lower body\", 1, 1)\nplot_image_samples(image_article_df, \"Garment Full body\", 1, 1)\nplot_image_samples(image_article_df, \"Accessories\", 1, 1)\nplot_image_samples(image_article_df, \"Swimwear\", 1, 1)\nplot_image_samples(image_article_df, \"Underwear\", 1, 1)","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:47:41.111022Z","iopub.execute_input":"2023-09-04T16:47:41.111439Z","iopub.status.idle":"2023-09-04T16:47:44.422096Z","shell.execute_reply.started":"2023-09-04T16:47:41.111406Z","shell.execute_reply":"2023-09-04T16:47:44.420565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tarih ve müşteri bazında aynı gün içinde yapılan işlemleri birleştirme ve toplamda kaç birleştirilmiş işlem var hesaplama\n\n\ndf_trs['t_dat'] = pd.to_datetime(df_trs['t_dat'])\n\ndf_trs['article_id'] = df_trs['article_id'].astype(str)\n\ncombined_transactions = df_trs.groupby(['customer_id', df_trs['t_dat'].dt.date])['article_id'].apply(lambda x: ', '.join(x)).reset_index()\n\ntotal_combined_transaction_count = len(combined_transactions)\n\nprint(\"Total Combined Transaction Count:\", total_combined_transaction_count)","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:47:44.424527Z","iopub.execute_input":"2023-09-04T16:47:44.425086Z","iopub.status.idle":"2023-09-04T16:53:47.133558Z","shell.execute_reply.started":"2023-09-04T16:47:44.425018Z","shell.execute_reply":"2023-09-04T16:53:47.12954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Toplam işlem sayısı bazında müşterileri sayma ve en yüksek işlem sayısına sahip müşteriyi bulma\n\ntransaction_counts = combined_transactions['customer_id'].value_counts()\n\nmax_transaction_customer = transaction_counts.idxmax()\nmax_transaction_count = transaction_counts.max()\n\nprint(f\"En yuksek sayida alisveris yapan musterinin numarasi {max_transaction_customer} \"\n      f\"toplam satis sayisi {max_transaction_count} ile\")","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:53:47.139088Z","iopub.execute_input":"2023-09-04T16:53:47.140239Z","iopub.status.idle":"2023-09-04T16:53:53.146523Z","shell.execute_reply.started":"2023-09-04T16:53:47.140095Z","shell.execute_reply":"2023-09-04T16:53:53.145294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Aktif club member sayisi\n\nunique_active_members = df_cust[df_cust['club_member_status'] == 'ACTIVE']['customer_id'].nunique()\n\nprint(\"Number of unique customers with ACTIVE club member status:\", unique_active_members)","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:53:53.14824Z","iopub.execute_input":"2023-09-04T16:53:53.14943Z","iopub.status.idle":"2023-09-04T16:53:54.704349Z","shell.execute_reply.started":"2023-09-04T16:53:53.149389Z","shell.execute_reply":"2023-09-04T16:53:54.700018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---------------------------------------------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"# 2. MÜŞTERİ SEGMENTASYONU (RFM Tekniği ile)","metadata":{}},{"cell_type":"code","source":"!pip install lifetimes\n\nimport pandas as pd\nimport datetime as dt\nimport numpy as np\nimport warnings\n\nwarnings.simplefilter(action='ignore', category=Warning)\nimport matplotlib.pyplot as plt\nfrom lifetimes import BetaGeoFitter\nfrom lifetimes import GammaGammaFitter\nfrom lifetimes.plotting import plot_period_transactions\nfrom sklearn.preprocessing import MinMaxScaler\n\n\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)\npd.set_option('display.float_format', lambda x: '%.2f' % x)\npd.options.mode.chained_assignment = None\npd.set_option('display.width', 500)\npd.set_option('display.expand_frame_repr', False)","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:53:54.709559Z","iopub.execute_input":"2023-09-04T16:53:54.710792Z","iopub.status.idle":"2023-09-04T16:54:14.280539Z","shell.execute_reply.started":"2023-09-04T16:53:54.710702Z","shell.execute_reply":"2023-09-04T16:54:14.279013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# veri okutma\n\ndf_articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_ladies = df_articles[df_articles['index_group_name'] == 'Ladieswear'] \ndf_ladies['article_id'].nunique()\nladieswear_article_ids = df_ladies['article_id'].tolist()\n\n\n# 'Transactions_train' dosyasi okutma\ndf_trs = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\n\n\n# Ilgili article id leri iceren satirlari 'df_trs' veri setinden cekme\ndf_trs_all_season = df_trs[df_trs['article_id'].isin(ladieswear_article_ids)]\n\ndf_trs_all_season.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:54:14.283038Z","iopub.execute_input":"2023-09-04T16:54:14.28518Z","iopub.status.idle":"2023-09-04T16:55:44.884405Z","shell.execute_reply.started":"2023-09-04T16:54:14.285098Z","shell.execute_reply":"2023-09-04T16:55:44.882968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def hm_rfm(df):\n    \n    # rfm skorlarının hazırlanması \n    df['t_dat'] = pd.to_datetime(df['t_dat'])\n    df[\"t_dat\"].max()\n\n    today_date= df[\"t_dat\"].max() + dt.timedelta(days=2)\n    today_date  # Timestamp('2020-09-24 00:00:00')\n    \n    rfm = df.groupby(\"customer_id\").agg(recency=(\"t_dat\", lambda x: (today_date - x.max()).days),\n                                           frequency=(\"t_dat\", \"nunique\"),\n                                           monetary=(\"price\", \"sum\"))\n    \n    # ölçeklendirme\n    rfm[\"recency_score\"]= pd.qcut(rfm[\"recency\"], 5, labels=[5,4,3,2,1])\n    rfm[\"monetary_score\"]= pd.qcut(rfm[\"monetary\"], 5, labels=[1,2,3,4,5])\n    rfm[\"frequency_score\"]= pd.qcut(rfm[\"frequency\"].rank(method=\"first\"), 5, labels=[1,2,3,4,5])\n    \n    # rf score \n    rfm[\"RF_SCORE\"]= (rfm[\"recency_score\"].astype(str)+ rfm[\"frequency_score\"].astype(str))\n    \n    # segmenasyon\n    seg_map = {\n    r\"[1-2][1-2]\": \"hibernating\",\n    r\"[1-2][3-4]\": \"at_Risk\",\n    r\"[1-2]5\": \"cant_loose\",\n    r\"3[1-2]\": \"about_to_sleep\",\n    r\"33\": \"need_attention\",\n    r\"[3-4][4-5]\": \"loyal_customers\",\n    r\"41\": \"promising\",\n    r\"51\": \"new_customers\",\n    r\"[4-5][2-3]\": \"potential_loyalists\",\n    r\"5[4-5]\": \"champions\"}\n    \n    rfm[\"segment\"] = rfm[\"RF_SCORE\"].replace(seg_map, regex=True)\n    \n    # indexten kurtarma \n    rfm.reset_index(inplace=True)\n    \n    return rfm","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:55:44.893602Z","iopub.execute_input":"2023-09-04T16:55:44.89408Z","iopub.status.idle":"2023-09-04T16:55:44.908533Z","shell.execute_reply.started":"2023-09-04T16:55:44.894045Z","shell.execute_reply":"2023-09-04T16:55:44.907046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfm= hm_rfm(df_trs_all_season)\nrfm.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:55:44.910674Z","iopub.execute_input":"2023-09-04T16:55:44.911209Z","iopub.status.idle":"2023-09-04T16:59:26.806701Z","shell.execute_reply.started":"2023-09-04T16:55:44.911145Z","shell.execute_reply":"2023-09-04T16:59:26.805367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---------------------------------------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"# 3. SEPET ANALİZİ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\npd.set_option('display.max_columns', None)\npd.set_option('display.width', 500)\npd.set_option('display.expand_frame_repr', False)\nfrom mlxtend.frequent_patterns import apriori, association_rules\nfrom datetime import datetime\n\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:59:26.809294Z","iopub.execute_input":"2023-09-04T16:59:26.809703Z","iopub.status.idle":"2023-09-04T16:59:26.831814Z","shell.execute_reply.started":"2023-09-04T16:59:26.809667Z","shell.execute_reply":"2023-09-04T16:59:26.830473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_ladies = df_articles[df_articles['index_group_name'] == 'Ladieswear'] \nladieswear_article_ids = df_ladies['article_id'].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:59:26.834019Z","iopub.execute_input":"2023-09-04T16:59:26.834563Z","iopub.status.idle":"2023-09-04T16:59:27.895566Z","shell.execute_reply.started":"2023-09-04T16:59:26.834514Z","shell.execute_reply":"2023-09-04T16:59:27.893877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def arl_data_preparation():\n    # Articles dosyasi okutma\n    df_articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\n    df_ladies = df_articles[df_articles['index_group_name'] == 'Ladieswear'] \n    ladieswear_article_ids = df_ladies['article_id'].tolist()\n\n    # 'Transactions_train' dosyasi okutma\n    df_trs = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\n\n    # Son sezonu secme\n    df_trs['t_dat'] = pd.to_datetime(df_trs['t_dat'])\n    cutoff_date = datetime(2020, 3, 21)\n    df_trs = df_trs[df_trs['t_dat'] > cutoff_date]\n\n\n    # Ilgili article id leri iceren satirlari 'df_trs' veri setinden cekme\n    filtered_data = df_trs[df_trs['article_id'].isin(ladieswear_article_ids)]\n\n\n    # \"article_id\" değeri 10 veya daha fazla olan satırları filtreleyin\n    filtered_data = filtered_data[filtered_data['article_id'].map(filtered_data['article_id'].value_counts()) >= 10]\n    \n    # sepet_id oluşturma \n    filtered_data['sepet_id'] = filtered_data['customer_id'].astype(str) + '_' + filtered_data['t_dat'].astype(str)\n    multiple_product_baskets = filtered_data.groupby('sepet_id').filter(lambda x: len(x) >= 10)\n    print(f\"Sepet Sayısı: {multiple_product_baskets['sepet_id'].nunique()}, Ürün Sayısı: {multiple_product_baskets['article_id'].nunique()}\")\n    result_df = filtered_data[filtered_data['sepet_id'].isin(multiple_product_baskets['sepet_id'].unique())]\n    \n    # Pivot table olusturma\n    basket = result_df.groupby(['sepet_id', 'article_id'])['article_id'].count().unstack().fillna(0).applymap(lambda x: 1 if x > 0 else 0).astype(bool)\n    \n    \n    # Frequent itemsets\n    frequent_itemsets = apriori(basket,\n                            min_support=0.0037,\n                            use_colnames=True)\n\n    frequent_itemsets.sort_values(\"support\", ascending=False)\n\n\n    rules = association_rules(frequent_itemsets,\n                          metric=\"lift\",\n                          min_threshold=1)\n    \n    return rules","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:59:27.897754Z","iopub.execute_input":"2023-09-04T16:59:27.898194Z","iopub.status.idle":"2023-09-04T16:59:27.913047Z","shell.execute_reply.started":"2023-09-04T16:59:27.898143Z","shell.execute_reply":"2023-09-04T16:59:27.911538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rules= arl_data_preparation()\nrules.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-04T16:59:27.915107Z","iopub.execute_input":"2023-09-04T16:59:27.91571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def arl_recommender(rules_df, main_article_id, rec_count=5):\n    sorted_rules = rules_df.sort_values(\"lift\", ascending=False)\n    recommendation_list = []\n    for i, antecedents_set in enumerate(sorted_rules[\"antecedents\"]):\n        if main_article_id in antecedents_set:\n            recommendation_list.append(list(sorted_rules.iloc[i][\"consequents\"])[0])\n\n    return recommendation_list[:rec_count]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_article_recommendations(rules, main_article_id):\n    recommended_articles = arl_recommender(rules, main_article_id)\n    print(recommended_articles)\n    \n    cols = (len(recommended_articles) + 1)\n    rows = (len(recommended_articles) + cols - 1) // cols\n    \n    _df = df_ladies[df_ladies['article_id'].isin(recommended_articles)]\n    article_ids = _df.article_id.values[0:cols*rows]\n    \n    plt.figure(figsize=(2 + 3 * cols, 2 + 4 * rows))\n    for i in range(cols * rows):\n        plt.subplot(rows, cols, i + 1)\n        plt.axis('off')\n        \n        if i == 0:\n            article_id = (\"0\" + str(main_article_id))[-10:]\n            plt.title(f\"Main Article {article_id}\")\n        else:\n            article_id = (\"0\" + str(article_ids[i-1]))[-10:]\n            plt.title(f\"Recommended Article {article_id}\")\n        \n        image = Image.open(f\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/{article_id[:3]}/{article_id}.jpg\")\n        plt.imshow(image)\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_article_recommendations(rules, 599580055)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"--------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"## **4. İŞBİRLİKÇİ FİLTRELEME**\n  ##   4.1 KULLANICI TABANLI ÖNERİ SİSTEMİ\n  ##   4.2 ÜRÜN TABANLI ÖNERİ SİSTEMİ","metadata":{}},{"cell_type":"markdown","source":"-----------------------------------------------------------------------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"code","source":"# pivot table oluşturma \ndf1 = filtered_transactions.groupby(['customer_id', 'article_id'])['article_id'].count().unstack().fillna(0).applymap(lambda x: 1 if x > 0 else 0).astype(int)\ndf = filtered_transactions.pivot_table(index='customer_id', columns='article_id', values='article_counts_label', fill_value=0)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-05T18:59:13.230021Z","iopub.execute_input":"2023-09-05T18:59:13.230498Z","iopub.status.idle":"2023-09-05T19:01:07.380094Z","shell.execute_reply.started":"2023-09-05T18:59:13.230458Z","shell.execute_reply":"2023-09-05T19:01:07.378927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4.1 KULLANICI TABANLI ÖNERİ SİSTEMİ","metadata":{}},{"cell_type":"code","source":"# kütüphaneler\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom sklearn.metrics.pairwise import cosine_similarity\nfrom datetime import datetime\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2023-09-05T18:52:32.011762Z","iopub.execute_input":"2023-09-05T18:52:32.012728Z","iopub.status.idle":"2023-09-05T18:52:32.55163Z","shell.execute_reply.started":"2023-09-05T18:52:32.012677Z","shell.execute_reply":"2023-09-05T18:52:32.550168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# veriyi okutma \n\ndf_articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_trs = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-05T16:28:32.244583Z","iopub.execute_input":"2023-09-05T16:28:32.245006Z","iopub.status.idle":"2023-09-05T16:30:04.297542Z","shell.execute_reply.started":"2023-09-05T16:28:32.244976Z","shell.execute_reply":"2023-09-05T16:30:04.296557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = '/kaggle/input/h-and-m-personalized-fashion-recommendations/images'","metadata":{"execution":{"iopub.status.busy":"2023-09-05T16:30:14.03936Z","iopub.execute_input":"2023-09-05T16:30:14.039861Z","iopub.status.idle":"2023-09-05T16:30:14.04787Z","shell.execute_reply.started":"2023-09-05T16:30:14.039816Z","shell.execute_reply":"2023-09-05T16:30:14.04656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def user_based_recommendation(df_trs,df_articles, customer_id):\n    \n    # kadın giyim filtreleme\n    df_ladies = df_articles[df_articles[\"index_group_name\"]==\"Ladieswear\"]\n    ladieswear_article_ids = df_ladies['article_id'].tolist()\n\n    # sezon seçme \n    df_trs['t_dat'] = pd.to_datetime(df_trs['t_dat'])\n    cutoff_date = datetime(2020, 3, 21)\n    df_trs = df_trs[df_trs['t_dat'] > cutoff_date]\n\n    # Ilgili article id leri iceren satirlari 'df_trs' veri setinden cekme\n    ladies_transactions = df_trs[df_trs['article_id'].isin(ladieswear_article_ids)]\n\n    # ladies_transactions veri setini istenilen başlıkları filtreleme \n    ladies_transactions = ladies_transactions[[\"customer_id\",\"article_id\",\"price\"]]\n    article_counts = pd.DataFrame(ladies_transactions.value_counts(\"article_id\"))\n\n    # ladies_transactions ile article_counts verilerini birleştirme \n    merged_df = ladies_transactions.merge(article_counts, on='article_id', how='left')\n    merged_df.columns = [\"customer_id\",\"article_id\",\"price\",\"article_counts\"]\n\n    # total value hesaplama ve 1-100 arasında ölçeklendirme \n    merged_df[\"total_value\"] = merged_df[\"price\"] * merged_df[\"article_counts\"]\n    merged_df = merged_df[merged_df[\"total_value\"] > 5]\n    merged_df['article_counts_label'] = pd.qcut(merged_df['total_value'], q=100, labels=range(1, 101))\n\n    # 30 adetten az ürün alan müşterilei filtreleme \n    customer_total_products = ladies_transactions.groupby('customer_id')['article_id'].count().sort_values(ascending=False)\n    less_than_30 = customer_total_products[customer_total_products < 30]\n    less_than_30_cust = less_than_30.index.tolist()\n    filtered_transactions = ladies_transactions[~ladies_transactions['customer_id'].isin(less_than_30_cust)]\n    filtered_transactions = pd.merge(filtered_transactions, merged_df[['customer_id', 'article_id', 'article_counts_label']], on=['customer_id', 'article_id'], how='inner')\n    filtered_transactions[\"article_counts_label\"] = merged_df[\"article_counts_label\"].astype(int)\n\n    # kullanıcı/ürün matrisini hazırlama \n    df1 = filtered_transactions.groupby(['customer_id', 'article_id'])['article_id'].count().unstack().fillna(0).applymap(lambda x: 1 if x > 0 else 0).astype(int)\n    matris = filtered_transactions.pivot_table(index='customer_id', columns='article_id', values='article_counts_label', fill_value=0)\n    matris = matris.reset_index()\n    matris.set_index(\"customer_id\", inplace=True)\n    \n    # Belirli müşterinin satın aldığı ürünleri liste olarak alın\n    true_articles = matris.columns[matris.iloc[matris.index == customer_id].values[0] > 0].tolist()\n    \n    # Tüm müşteriler için bu ürünleri içeren DataFrame'i oluşturma\n    common_articles_df = matris[true_articles]\n    articles_count = common_articles_df.T.notnull().sum()\n    \n    # Hesaplanan ürün sayılarını bir DataFrame'e dönüştürme\n    articles_count = articles_count.reset_index()\n    articles_count.columns = [\"customer_id\", \"articles_count\"]\n    \n    # Ortak ürünleri alan müşterilerin kimliklerini listeye alma\n    common_customer_df = articles_count[\"customer_id\"].tolist()\n    \n    # Ortak müşterileri ve belirli müşteriyi içeren son DataFrame'i oluşturma\n    final_df = pd.concat([matris[matris.index.isin(common_customer_df)], matris[matris.index == customer_id]])\n    \n    # Müşteri kimliklerini alma\n    customer_ids = final_df.index\n    \n    # DataFrame'in değerlerini bir NumPy dizisine dönüştürme\n    data = final_df.values\n    \n    # Cosine benzerliğini hesaplama\n    cosine_sim = cosine_similarity(data)\n    cosine_sim_df = pd.DataFrame(cosine_sim, index=customer_ids, columns=customer_ids)\n    \n    # Belirli müşterinin benzerlik skorlarını çekme/sıralama\n    cosine_random = cosine_sim_df.loc[customer_id, :]\n    cosine_sorted = cosine_random.iloc[0, :].sort_values(ascending=False)\n    cosine_sorted = cosine_sorted.reset_index()\n    cosine_sorted.columns = [\"customer_id\", \"cosine\"]\n    cosine_sorted = cosine_sorted[cosine_sorted[\"customer_id\"] != customer_id]  # Tavsiye verilecek müşteriyi çıkar \n    \n    # En iyi benzer müşterileri seçme\n    top_users = cosine_sorted\n    top_users_ratings = top_users.merge(filtered_transactions[[\"customer_id\", \"article_id\", \"article_counts_label\"]], how=\"inner\")\n    \n    # Ağırlıklı puan hesaplama (benzerlik skoru * ürün satın alma sayısı)\n    top_users_ratings[\"weighted_rating\"] = top_users_ratings[\"cosine\"] * top_users_ratings[\"article_counts_label\"]\n    recommendation_df = top_users_ratings.groupby(\"article_id\").agg({\"weighted_rating\": \"mean\"})\n    recommendation_df = recommendation_df.reset_index()\n    \n    # Ürün tavsiyelerini alın\n    recommends = recommendation_df[\"article_id\"].head(10).tolist()\n    \n    print(\"Satın Alınan Ürünler: \" + \".\".join(map(str, true_articles)))\n    print(\"---------------------------------------------\")\n    print(\"Tavsiye Edilen Ürünler: \" + \".\".join(map(str, recommends)))\n    \n    return true_articles, recommends","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true_articles, recommends = user_based_recommendation(df_trs,df_articles, \"0006d3ff0caf0cb4d4e0615ee5cb7d268622364d483335bf21fbf296e785e282\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_content_based_recommendations(599580055, cosine_sim, df_ladies)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_user_based_recommendations(true_articles, recommended_articles, image_path):\n    num_true_articles = len(true_articles)\n    num_recommended_articles = len(recommended_articles)\n    \n    cols = 15\n    \n    # Display True Articles\n    rows_true = (num_true_articles + cols - 1) // cols\n    plt.figure(figsize=(4 * cols, 6 * rows_true))\n        \n    for i, article_id in enumerate(true_articles, start=1):\n        plt.subplot(rows_true, cols, i)\n        plt.axis('off')\n        if i == 1:\n            # Ana ürünü ilk sırada gösterin\n            article_id_str = (\"0\" + str(article_id))[-10:]\n            plt.title(f\"Main Article {article_id_str}\")\n        else:\n            # Önerilen ürünleri sırayla gösterin\n            article_id_str = (\"0\" + str(article_id))[-10:]\n            plt.title(f\"True Article {article_id_str}\")\n\n        image = Image.open(f\"{image_path}/{article_id_str[:3]}/{article_id_str}.jpg\")\n        plt.imshow(image)\n        \n    plt.tight_layout()\n    plt.show()\n    \n    # Display Recommended Articles\n    rows_recommended = (num_recommended_articles + cols - 1) // cols\n    plt.figure(figsize=(4 * cols, 6 * rows_recommended))\n    \n    for i, article_id in enumerate(recommended_articles, start=1):\n        plt.subplot(rows_recommended, cols, i)\n        plt.axis('off')\n        if i == 1:\n            # Ana ürünü ilk sırada gösterin\n            article_id_str = (\"0\" + str(article_id))[-10:]\n            plt.title(f\"Main Article {article_id_str}\")\n        else:\n            # Önerilen ürünleri sırayla gösterin\n            article_id_str = (\"0\" + str(article_id))[-10:]\n            plt.title(f\"Recommended Article {article_id_str}\")\n\n        image = Image.open(f\"{image_path}/{article_id_str[:3]}/{article_id_str}.jpg\")\n        plt.imshow(image)\n        \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_user_based_recommendations(true_articles, recommends, image_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"------------------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"# 4.2 ÜRÜN TABANLI ÖNERİ SİSTEMİ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom datetime import datetime\nfrom sklearn.metrics.pairwise import cosine_similarity\n\n# # Set display options\npd.set_option('display.max_columns', None)\npd.set_option('display.width', 500)\npd.set_option('display.expand_frame_repr', False)\n\nimport warnings\nwarnings.simplefilter(\"ignore\")\n\nfrom PIL import Image\nimport matplotlib.pyplot as plt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Veri seti okutma \n\ndf_articles = pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ndf_ladies = df_articles[df_articles[\"index_group_name\"]==\"Ladieswear\"]\nladieswear_article_ids = df_ladies['article_id'].tolist()\ndf_trs = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ıtem_based_data_preparation(df_trs):\n    \n    # Son sezonu secme\n    df_trs['t_dat'] = pd.to_datetime(df_trs['t_dat'])\n    cutoff_date = datetime(2020, 3, 21)\n    df_trs = df_trs[df_trs['t_dat'] > cutoff_date]\n    \n    # Ilgili article id leri iceren satirlari 'df_trs' veri setinden cekme\n    ladies_transactions = df_trs[df_trs['article_id'].isin(ladieswear_article_ids)]\n    ladies_transactions = ladies_transactions[[\"customer_id\",\"article_id\"]]\n    article_counts = pd.DataFrame(ladies_transactions.value_counts(\"article_id\"))\n    merged_df = ladies_transactions.merge(article_counts, on='article_id', how='left')\n    merged_df.columns = [\"customer_id\",\"article_id\",\"article_counts\"]\n    merged_df = merged_df[merged_df[\"article_counts\"] > 100]\n    merged_df['article_counts_label'] = pd.cut(merged_df['article_counts'], bins=100, labels=range(1, 101))\n    ladies_transactions=merged_df\n    customer_total_products = ladies_transactions.groupby('customer_id')['article_id'].count().sort_values(ascending=False)\n    less_than_30 = customer_total_products[customer_total_products < 30]\n    less_than_30_cust = less_than_30.index.tolist()\n    \n    # 30 dan az ürün alan müşterileri veri setinden çıkarma\n    \n    filtered_transactions = ladies_transactions[~ladies_transactions['customer_id'].isin(less_than_30_cust)]\n    \n    from sklearn.metrics.pairwise import cosine_similarity\n    \n    filtered_transactions[\"article_counts_label\"] = filtered_transactions[\"article_counts_label\"].astype(int)\n\n    df = filtered_transactions.groupby(['customer_id', 'article_id'])['article_id'].count().unstack().fillna(0).applymap(lambda x: 1 if x > 0 else 0).astype(int)\n\n    return df\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= ıtem_based_data_preparation(df_trs)\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def item_based_recommender(id_name, df):\n       \n    id_vector = df[id_name].values.reshape(1, -1)\n    similarities = cosine_similarity(id_vector, df.T)  # Veriyi transpoze ederek sütunları ile satırları karşılaştırma\n    similar_items = pd.DataFrame(similarities, columns=df.columns, index=['Similarity']).T\n    similar_items = similar_items.sort_values(by='Similarity', ascending=False).head(10)\n    return similar_items","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_name = pd.Series(df.columns).sample(1, random_state=2).values[0]\nid_name","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_based_recommender(916866001, df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_item_based_recommendations(df, selected_item_id, num_recommendations=6):\n    \n    image_path= \"../input/h-and-m-personalized-fashion-recommendations\"\n    \n    # Öneri sonuçlarını almak için bir ürün ID'si\n    selected_item_id = selected_item_id\n\n    # Öneri sonuçları(örnek olarak rastgele öneriler)\n    recommendations = item_based_recommender(selected_item_id, df)\n\n    # Öneri sonuçlarından önerilen ürün ID'leri\n    recommended_articles = recommendations.index.tolist()\n\n    # Ana ürünü recommended_articles listesinden çıkarma\n    recommended_articles.remove(selected_item_id)\n\n    # Öneri sonuçlarını belirtilen sayıda ürün ile sınırlama\n    recommended_articles = recommended_articles[:num_recommendations]\n\n    # Ana ürünün görselini gösterme\n    plt.figure(figsize=(6, 6))\n    plt.subplot(1, 1, 1)\n    plt.axis('off')\n    main_article_id_str = (\"0\" + str(selected_item_id))[-10:]\n    plt.title(f\"Ana Ürün {main_article_id_str}\")\n    main_image = Image.open(f\"{image_path}/images/{main_article_id_str[:3]}/{main_article_id_str}.jpg\")\n    plt.imshow(main_image)\n\n    # Önerilen ürünleri görselleştirme\n    cols = 2\n    rows = (len(recommended_articles) + cols - 1) // cols\n\n    plt.figure(figsize=(6 * cols, 6 * rows))\n    for i, article_id in enumerate(recommended_articles):\n        plt.subplot(rows, cols, i + 1)\n        plt.axis('off')\n\n        article_id_str = (\"0\" + str(article_id))[-10:]\n        plt.title(f\"Önerilen Ürün {article_id_str}\")\n\n        image = Image.open(f\"{image_path}/images/{article_id_str[:10]}/{article_id_str}.jpg\")\n        plt.imshow(image)\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_item_based_recommendations(df, 916866001, num_recommendations=6)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"## **5. İÇERİK TABANLI FİLTRELEME**\n##  5.1 METİN TABANLI ÖNERİ SİSTEMİ\n##  5.2 GÖRÜNTÜ TABANI ÖNERİ SİSTEMİ (CNN ile)","metadata":{}},{"cell_type":"markdown","source":"--------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"# 5.1 METİN TABANLI ÖNERİ SİSTEMİ","metadata":{}},{"cell_type":"code","source":"# Importing libraries\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport re\nfrom nltk.corpus import stopwords\nfrom nltk.stem import PorterStemmer\nfrom nltk.tokenize import word_tokenize\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics.pairwise import cosine_similarity\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport random\n\n# # Set display options\npd.set_option('display.max_columns', None)\npd.set_option('display.width', 500)\npd.set_option('display.expand_frame_repr', False)\nimport warnings\nwarnings.simplefilter(\"ignore\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Veriyi okutma \n\ndf_articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_ladies = df_articles[df_articles['index_group_name'] == 'Ladieswear']\ndf_ladies = df_ladies.reset_index(drop=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_text(dataframe, detail_col ):\n    # Eksik veriyi tanımlama/silme \n    missing_count = dataframe[detail_col].isnull().sum()\n    dataframe = dataframe.dropna(subset=[detail_col])\n    dataframe = dataframe.reset_index(drop=True)\n    df = dataframe.select_dtypes(include=['object'])\n\n    documents = df.apply(' '.join, axis=1)   # Tüm metin sütunlarını birleştirin\n    documents = documents.str.lower() # Her bir dizeyi küçük harfe dönüştürme\n    documents = documents.str.replace(r'[^\\w\\s]', ' ', regex=True) # Özel karakterleri boşlukla değiştirme\n    tokens = documents.str.split() # Kelimeleri bölme\n    cleaned_documents = tokens.apply(lambda token_list: \" \".join(token_list)) # Her bir belgenin kelimelerini birleştirme\n    filtered_tokens = tokens.apply(lambda token_list: [word for word in token_list if word not in stopwords.words('english')])  # Stop kelimeleri filtreleme\n    \n    return cleaned_documents","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"documents =preprocess_text(df_ladies, \"detail_desc\")\nprint(documents)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cosine_sim(text):\n    \n    # TF-IDF vectorization\n    tfidf = TfidfVectorizer()\n    \n    # Convert the document collection to TF-IDF vectors\n    tfidf_matrix = tfidf.fit_transform(text)\n    \n    # Calculating the similarity between TF-IDF vectors of documents using the cosine similarity metric\n    cosine_sim = cosine_similarity(tfidf_matrix,tfidf_matrix)\n    \n    return cosine_sim","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cosine_sim= cosine_sim(documents)\nprint(cosine_sim)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def content_based_recommender(article_id, cosine_sim, dataframe):\n    indices = pd.Series(dataframe.index, index=dataframe['article_id'])\n    indices = indices[~indices.index.duplicated(keep='last')]\n    product_index = indices[article_id]\n    similarity_scores = pd.DataFrame(cosine_sim[product_index], columns=[\"score\"])\n    product_indices = similarity_scores.sort_values(\"score\", ascending=False).index[1:6]\n\n    return dataframe['article_id'].iloc[product_indices]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommended_articles = content_based_recommender(599580055, cosine_sim, df_ladies)\nrecommended_articles","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_content_based_recommendations(article_id, cosine_sim, dataframe):\n    # ürün indexlerini ve benzerlik puanlarını içeren bir DataFrame oluşturun\n    indices = pd.Series(dataframe.index, index=dataframe['article_id'])\n    indices = indices[~indices.index.duplicated(keep='last')]\n    product_index = indices[article_id]\n    similarity_scores = pd.DataFrame(cosine_sim[product_index], columns=[\"score\"])\n    product_indices = similarity_scores.sort_values(\"score\", ascending=False).index[1:6]\n\n    # Önerilen ürünün article_id'lerini alın\n    recommended_articles = dataframe['article_id'].iloc[product_indices]\n\n    image_path = \"/kaggle/input/h-and-m-personalized-fashion-recommendations\"\n\n    # Önerilen ürünleri görsel olarak düzenle\n    cols = 2\n    rows = (len(recommended_articles) + cols - 1) // cols\n\n    # Önerilen ürünlerin veri çerçevesini filtreleyin\n    _df = df_ladies[df_ladies['article_id'].isin(recommended_articles)]\n    article_ids = _df.article_id.values[0:cols*rows]\n\n    # Bir altlık ve görsel için yer açın\n    plt.figure(figsize=(2 + 3 * cols, 2 + 4 * rows))\n    for i in range(cols * rows):\n        plt.subplot(rows, cols, i + 1)\n        plt.axis('off')\n\n        if i == 0:\n            # Ana ürünü ilk sırada gösterin\n            article_id = (\"0\" + str(article_id))[-10:]\n            plt.title(f\"Main Article {article_id}\")\n        else:\n            # Önerilen ürünleri sırayla gösterin\n            article_id = (\"0\" + str(article_ids[i-1]))[-10:]\n            plt.title(f\"Recommended Article {article_id}\")\n\n        image = Image.open(f\"{image_path}/images/{article_id[:3]}/{article_id}.jpg\")\n        plt.imshow(image)\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}