{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\"\"\"\n- 将包括article_id、category_id在内的所有类别转换为以0为索引的顺序号（带有_idx的列被添加）。\n- 将只有None、1的类别转换为0、1（列被覆盖）。\n- 将只有1，2的类别转换为0，1（列被覆盖）。\n\"\"\"\nimport shutil\nfrom pathlib import Path\nfrom typing import Any\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-13T12:59:03.272333Z","iopub.execute_input":"2023-11-13T12:59:03.273657Z","iopub.status.idle":"2023-11-13T12:59:03.766356Z","shell.execute_reply.started":"2023-11-13T12:59:03.273616Z","shell.execute_reply":"2023-11-13T12:59:03.765194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 数据格式\nARTICLES_ORIGINAL = {\n    'article_id': 'object',\n    'product_code': 'int64',\n    'prod_name': 'object',\n    'product_type_no': 'int64',\n    'product_type_name': 'object',\n    'product_group_name': 'object',\n    'graphical_appearance_no': 'int64',\n    'graphical_appearance_name': 'object',\n    'colour_group_code': 'int64',\n    'colour_group_name': 'object',\n    'perceived_colour_value_id': 'int64',\n    'perceived_colour_value_name': 'object',\n    'perceived_colour_master_id': 'int64',\n    'perceived_colour_master_name': 'object',\n    'department_no': 'int64',\n    'department_name': 'object',\n    'index_code': 'object',\n    'index_name': 'object',\n    'index_group_no': 'int64',\n    'index_group_name': 'object',\n    'section_no': 'int64',\n    'section_name': 'object',\n    'garment_group_no': 'int64',\n    'garment_group_name': 'object',\n    'detail_desc': 'object',\n}\n\nCUSTOMERS_ORIGINAL = {\n    'customer_id': 'object',\n    'FN': 'float64',\n    'Active': 'float64',\n    'club_member_status': 'object',\n    'fashion_news_frequency': 'object',\n    'age': 'float64',\n    'postal_code': 'object',\n}\n\nTRANSACTIONS_ORIGINAL = {\n    'customer_id': 'object',\n    'article_id': 'object',\n    'price': 'float64',\n    'sales_channel_id': 'int64',\n}\n","metadata":{"execution":{"iopub.status.busy":"2023-11-13T12:59:03.768147Z","iopub.execute_input":"2023-11-13T12:59:03.768616Z","iopub.status.idle":"2023-11-13T12:59:03.776601Z","shell.execute_reply.started":"2023-11-13T12:59:03.768583Z","shell.execute_reply":"2023-11-13T12:59:03.775372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/input/h-and-m-personalized-fashion-recommendations'\narticles = pd.read_csv(f'{data_dir}/articles.csv', dtype=ARTICLES_ORIGINAL) # 读取 articles.csv\ncustomers = pd.read_csv(f'{data_dir}/customers.csv', dtype=CUSTOMERS_ORIGINAL) # 读取 customers.csv\n# 读取transactions_train\ntransactions = pd.read_csv(\n    f'{data_dir}/transactions_train.csv',\n    dtype=TRANSACTIONS_ORIGINAL,\n    parse_dates=['t_dat']# 解析日期特征\n)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T12:59:05.974613Z","iopub.execute_input":"2023-11-13T12:59:05.974998Z","iopub.status.idle":"2023-11-13T13:00:29.743047Z","shell.execute_reply.started":"2023-11-13T12:59:05.974966Z","shell.execute_reply":"2023-11-13T13:00:29.741987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.iloc[:10]","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:00:29.744739Z","iopub.execute_input":"2023-11-13T13:00:29.745156Z","iopub.status.idle":"2023-11-13T13:00:29.779101Z","shell.execute_reply.started":"2023-11-13T13:00:29.745125Z","shell.execute_reply":"2023-11-13T13:00:29.778084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers.iloc[:10]","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:00:29.780666Z","iopub.execute_input":"2023-11-13T13:00:29.781245Z","iopub.status.idle":"2023-11-13T13:00:29.798370Z","shell.execute_reply.started":"2023-11-13T13:00:29.781208Z","shell.execute_reply":"2023-11-13T13:00:29.796812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions.iloc[:10]","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:00:29.800990Z","iopub.execute_input":"2023-11-13T13:00:29.801393Z","iopub.status.idle":"2023-11-13T13:00:29.816963Z","shell.execute_reply.started":"2023-11-13T13:00:29.801360Z","shell.execute_reply":"2023-11-13T13:00:29.815703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# 假设你的DataFrame是df，你想要查询的列是'column_name'\nmost_frequent_item = transactions['customer_id'].value_counts().idxmax()\nprint(most_frequent_item)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:00:29.818220Z","iopub.execute_input":"2023-11-13T13:00:29.818649Z","iopub.status.idle":"2023-11-13T13:00:37.214231Z","shell.execute_reply.started":"2023-11-13T13:00:29.818618Z","shell.execute_reply":"2023-11-13T13:00:37.213035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# different_trans= transactions['customer_id'].value_counts().head()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-13T11:06:39.623359Z","iopub.execute_input":"2023-11-13T11:06:39.623692Z","iopub.status.idle":"2023-11-13T11:06:51.241857Z","shell.execute_reply.started":"2023-11-13T11:06:39.623649Z","shell.execute_reply":"2023-11-13T11:06:51.239865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# most_frequent_item = different_trans.index[2]","metadata":{"execution":{"iopub.status.busy":"2023-11-13T11:10:44.151017Z","iopub.execute_input":"2023-11-13T11:10:44.151469Z","iopub.status.idle":"2023-11-13T11:10:44.157962Z","shell.execute_reply.started":"2023-11-13T11:10:44.151434Z","shell.execute_reply":"2023-11-13T11:10:44.156446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_user_item = transactions[transactions['customer_id']==most_frequent_item]\nmost_user_item","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:00:37.215769Z","iopub.execute_input":"2023-11-13T13:00:37.216123Z","iopub.status.idle":"2023-11-13T13:00:39.737068Z","shell.execute_reply.started":"2023-11-13T13:00:37.216092Z","shell.execute_reply":"2023-11-13T13:00:39.735874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_user_item = most_user_item[most_user_item['article_id'] <= \"0392168010\"]\nmost_user_item","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:00:49.195607Z","iopub.execute_input":"2023-11-13T13:00:49.195978Z","iopub.status.idle":"2023-11-13T13:00:49.214682Z","shell.execute_reply.started":"2023-11-13T13:00:49.195951Z","shell.execute_reply":"2023-11-13T13:00:49.213507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# 假设你的DataFrame是df\ndf = most_user_item\n\n# 获取文件夹a下的所有子文件夹\nsubfolders = [f.path for f in os.scandir('/kaggle/input/h-and-m-personalized-fashion-recommendations/images') if f.is_dir()]\n\n# 初始化一个空列表来存储所有的article_id\narticle_ids = []\n\n# 遍历每个子文件夹\nfor subfolder in subfolders:\n    # 获取子文件夹下的所有jpg文件\n    jpg_files = [f for f in os.listdir(subfolder) if f.endswith('.jpg')]\n    # 从jpg文件名中提取article_id并添加到列表中\n    article_ids.extend([f[:-4] for f in jpg_files])  # 去掉.jpg后缀\n\n# 筛选DataFrame\nfiltered_df = df[df['article_id'].isin(article_ids)]\n\nfiltered_df","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:01:16.823638Z","iopub.execute_input":"2023-11-13T13:01:16.824257Z","iopub.status.idle":"2023-11-13T13:01:31.386003Z","shell.execute_reply.started":"2023-11-13T13:01:16.824223Z","shell.execute_reply":"2023-11-13T13:01:31.383963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_df = filtered_df.drop('customer_id', axis=1)\nfiltered_df","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:05:48.957322Z","iopub.execute_input":"2023-11-13T13:05:48.957700Z","iopub.status.idle":"2023-11-13T13:05:49.029256Z","shell.execute_reply.started":"2023-11-13T13:05:48.957661Z","shell.execute_reply":"2023-11-13T13:05:49.027526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:06:26.104293Z","iopub.execute_input":"2023-11-13T13:06:26.104707Z","iopub.status.idle":"2023-11-13T13:06:26.112419Z","shell.execute_reply.started":"2023-11-13T13:06:26.104674Z","shell.execute_reply":"2023-11-13T13:06:26.111297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_list = filtered_df['article_id'].to_list()\nitem_list","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:01:31.407728Z","iopub.execute_input":"2023-11-13T13:01:31.408108Z","iopub.status.idle":"2023-11-13T13:01:31.420292Z","shell.execute_reply.started":"2023-11-13T13:01:31.408077Z","shell.execute_reply":"2023-11-13T13:01:31.419047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\n# 假设你的字符串是string\n# string = \"0368038004\"\n\nfor index,string in enumerate(item_list):\n    # 构造文件路径\n    file_path = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/\" + string[:3] + '/' + string + '.jpg'\n\n    # 读取并显示图片\n    img = mpimg.imread(file_path)\n    imgplot = plt.imshow(img)\n    plt.show()\n    print(filtered_df.iloc[index].to_list())  ","metadata":{"execution":{"iopub.status.busy":"2023-11-13T13:06:53.035336Z","iopub.execute_input":"2023-11-13T13:06:53.035704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}