{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.11"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":308.215431,"end_time":"2025-06-11T13:42:16.824269","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-06-11T13:37:08.608838","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <div style=\"background-color: #7851A9; padding: 12px; font-size: 18px; color: #FFFFFF; border-left: 8px solid #C9A0DC; width: 100%;\"><b>🎩[Competiton : H&M Personalized Fashion Recommendations](https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations)</b></div>","metadata":{"papermill":{"duration":0.007962,"end_time":"2025-06-11T13:37:12.983656","exception":false,"start_time":"2025-06-11T13:37:12.975694","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# <div style=\"background-color: #7851A9; padding: 12px; font-size: 18px; color: #FFFFFF; border-left: 8px solid #C9A0DC; width: 100%;\"><b></b>✨🚀📖Introduction</div>","metadata":{"papermill":{"duration":0.006436,"end_time":"2025-06-11T13:37:12.996938","exception":false,"start_time":"2025-06-11T13:37:12.990502","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<div style=\"background-color: #7851A9; border-left: 8px solid #C9A0DC; padding: 20px; border-radius: 8px; font-size: 14px; color: #FFFFFF;\">\n  <h5 style=\"font-size: 16px; margin-bottom: 10px; background-color: #4B0082; padding: 8px; border-radius: 4px; color: #FFFFFF;\">\n    <strong>Executive Summary</strong> \n  </h5>\n  <ul>\n    <li>1:describe later</li> \n    <li>2:describe later</li> \n    <li>3:describe later</li> \n  </div>","metadata":{"papermill":{"duration":0.006371,"end_time":"2025-06-11T13:37:13.010064","exception":false,"start_time":"2025-06-11T13:37:13.003693","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<div style=\"background-color: #7851A9; border-left: 8px solid #C9A0DC; padding: 20px; border-radius: 8px; font-size: 14px; color: #FFFFFF;\">\n  <h5 style=\"font-size: 16px; margin-bottom: 10px; background-color: #4B0082; padding: 8px; border-radius: 4px; color: #FFFFFF;\">\n    <strong>Summary</strong> \n  </h5>\n  <ul>\n    <li>1:describe later</li> \n    <li>2:describe later</li> \n    <li>3:describe later</li> \n  </div>","metadata":{"papermill":{"duration":0.006424,"end_time":"2025-06-11T13:37:13.023129","exception":false,"start_time":"2025-06-11T13:37:13.016705","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# <div style=\"background-color: #7851A9; padding: 12px; font-size: 18px; color: #FFFFFF; border-left: 8px solid #C9A0DC; width: 100%;\"><b></b>01 :🧰 Importing Libraries</div>","metadata":{"papermill":{"duration":0.006795,"end_time":"2025-06-11T13:37:13.036454","exception":false,"start_time":"2025-06-11T13:37:13.029659","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Table Manipulation, CalculatinTg\nimport pandas as pd\nimport numpy as np\npd.set_option('display.max_columns', 100) # increase the maximum number of columns\n\n# Visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.cm as cm\nimport missingno as msno\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\nfrom statsmodels.tools.tools import add_constant\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import silhouette_score, silhouette_samples\n\n# Ignore all warnings\nimport warnings\nwarnings.simplefilter(\"ignore\")\n\n# Set seed for reproducibility\nnp.random.seed(42)","metadata":{"papermill":{"duration":3.905889,"end_time":"2025-06-11T13:37:16.949120","exception":false,"start_time":"2025-06-11T13:37:13.043231","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:37:43.556587Z","iopub.execute_input":"2025-07-09T13:37:43.557179Z","iopub.status.idle":"2025-07-09T13:37:45.133165Z","shell.execute_reply.started":"2025-07-09T13:37:43.557156Z","shell.execute_reply":"2025-07-09T13:37:45.132419Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"background-color: #7851A9; padding: 12px; font-size: 18px; color: #FFFFFF; border-left: 8px solid #C9A0DC; width: 100%;\"><b></b>02 :📥 Importing Datasets</div>\n\n<div style=\"background-color: #7851A9; border-left: 8px solid #C9A0DC; padding: 20px; border-radius: 8px; font-size: 14px; color: #FFFFFF;\">\n  <h5 style=\"font-size: 16px; margin-bottom: 10px; background-color: #4B0082; padding: 8px; border-radius: 4px; color: #FFFFFF;\">\n    <strong>Findings</strong> \n  </h5>\n    └ 1 : describe later<br>\n\n  <h5 style=\"font-size: 16px; margin-bottom: 10px; background-color: #4B0082; padding: 8px; border-radius: 4px; color: #FFFFFF;\">\n    <strong>Insights</strong> \n  </h5>\n    └ 1 : describe later<br>\n  </div>","metadata":{"papermill":{"duration":0.00627,"end_time":"2025-06-11T13:37:16.962338","exception":false,"start_time":"2025-06-11T13:37:16.956068","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df_articles  = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_customers = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\ndf_train     = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')","metadata":{"papermill":{"duration":1.441364,"end_time":"2025-06-11T13:37:18.410320","exception":false,"start_time":"2025-06-11T13:37:16.968956","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:37:45.321285Z","iopub.execute_input":"2025-07-09T13:37:45.322098Z","iopub.status.idle":"2025-07-09T13:39:03.043394Z","shell.execute_reply.started":"2025-07-09T13:37:45.322071Z","shell.execute_reply":"2025-07-09T13:39:03.042558Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"background-color: #7851A9; border-left: 8px solid #C9A0DC; padding: 20px; border-radius: 8px; font-size: 14px; color: #FFFFFF;\">\n  <h5 style=\"font-size: 16px; margin-bottom: 10px; background-color: #4B0082; padding: 8px; border-radius: 4px; color: #FFFFFF;\">\n    <strong>Column list:</strong> \n  </h5>\n    └df_articles(articles.csv)：unique products <br>\n    └article_id: A unique identifier for each product.<br>\n    └product_code: The product code of the item.<br>\n    └prod_name: Product name.<br>\n    └product_type_no: A numeric code that identifies the product type.<br>\n    └product_type_name: Product type name.<br>\n    <br>\n    └product_group_name: Product group name.<br>\n    └graphical_appearance_no: A numeric code that identifies a graphical appearance.<br>\n    └graphical_appearance_name: Graphic appearance name.<br>\n    └colour_group_code: A numeric code that identifies a color group.<br>\n    └colour_group_name: Color group name.<br>\n    <br>\n    └perceived_colour_value_id: A numeric ID that identifies the perceived value of a color (e.g. lightness).<br>\n    └perceived_colour_value_name: Perceived color value name.<br>\n    └perceived_colour_master_id: A numeric ID that identifies the master color of the perceived color.<br>\n    └perceived_colour_master_name: The master color name of the perceived color.<br>\n    └department_no: A numeric code that identifies the department.<br>\n    <br>\n    └department_name: Department name.<br>\n    └index_code: Index code.<br>\n    └index_name: Index name.<br>\n    └index_group_no: A numeric code that identifies an index group.<br>\n    └index_group_name: The index group name.<br>\n    <br>\n    └section_no: A numeric code that identifies the section.<br>\n    └section_name: Section name.<br>\n    └garment_group_no: A numeric code that identifies a clothing group.<br>\n    └garment_group_name: Clothing group name.<br>\n    └detail_desc: A detailed description of the product.<br>\n    <br>\n    <br>\n    <br>\n    └df_customers(customers.csv):unique customers. <br>\n    └customer_id: A unique identifier for each customer.<br>\n    └FN (Fashion News): Fashion newsletter subscription status.<br>\n    └Active: The active state of the customer.<br>\n    └club_member_status: Club membership status.<br>\n    └fashion_news_frequency: Delivery frequency of fashion news.<br>\n    └age: customer age.<br>\n    └postal_code: Customer postal code<br>\n    <br>\n    <br>\n    <br>\n    └ df_train(transactions_train.csv):transactions. <br>\n    └t_dat: The date the transaction took place.<br>\n    └customer_id: A unique identifier for the customer who performed the transaction.<br>\n    └article_id: A unique identifier for the traded instrument.<br>\n    └price: The selling price of the traded item.<br>\n    └sales_channel_id: Identifier of the sales channel through which the transaction took place.<br>\n</div>","metadata":{"papermill":{"duration":0.00717,"end_time":"2025-06-11T13:37:18.425279","exception":false,"start_time":"2025-06-11T13:37:18.418109","status":"completed"},"tags":[]}},{"cell_type":"code","source":"total_memory_bytes = df_articles.memory_usage(deep=True).sum()\nprint(f\"Total memory usage for the entire DataFrame: {total_memory_bytes} byte\")\nprint(f\"Total memory usage for the entire DataFrame: {total_memory_bytes / (1024**2):.2f} MB\") # megabyte display\nprint(f\"Total memory usage for the entire DataFrame: {total_memory_bytes / (1024**3):.2f} GB\")  # Gigabyte display\ndisplay(df_articles.info(memory_usage='deep'))\n\ndisplay(df_articles)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:39:03.672568Z","iopub.execute_input":"2025-07-09T13:39:03.673184Z","iopub.status.idle":"2025-07-09T13:39:04.282956Z","shell.execute_reply.started":"2025-07-09T13:39:03.673150Z","shell.execute_reply":"2025-07-09T13:39:04.282199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_memory_bytes = df_customers.memory_usage(deep=True).sum()\nprint(f\"Total memory usage for the entire DataFrame: {total_memory_bytes} byte\")\nprint(f\"Total memory usage for the entire DataFrame: {total_memory_bytes / (1024**2):.2f} MB\") # megabyte display\nprint(f\"Total memory usage for the entire DataFrame: {total_memory_bytes / (1024**3):.2f} GB\")  # Gigabyte display\ndisplay(df_customers.info(memory_usage='deep'))\n\ndisplay(df_customers)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:39:04.284530Z","iopub.execute_input":"2025-07-09T13:39:04.284772Z","iopub.status.idle":"2025-07-09T13:39:06.301586Z","shell.execute_reply.started":"2025-07-09T13:39:04.284755Z","shell.execute_reply":"2025-07-09T13:39:06.300776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_memory_bytes = df_train.memory_usage(deep=True).sum()\nprint(f\"Total memory usage for the entire DataFrame: {total_memory_bytes} byte\")\nprint(f\"Total memory usage for the entire DataFrame: {total_memory_bytes / (1024**2):.2f} MB\") # megabyte display\nprint(f\"Total memory usage for the entire DataFrame: {total_memory_bytes / (1024**3):.2f} GB\")  # Gigabyte display\ndisplay(df_train.info(memory_usage='deep'))\n\ndisplay(df_train)","metadata":{"papermill":{"duration":0.850642,"end_time":"2025-06-11T13:37:19.283145","exception":false,"start_time":"2025-06-11T13:37:18.432503","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:39:06.302674Z","iopub.execute_input":"2025-07-09T13:39:06.302925Z","iopub.status.idle":"2025-07-09T13:39:26.190409Z","shell.execute_reply.started":"2025-07-09T13:39:06.302907Z","shell.execute_reply":"2025-07-09T13:39:26.189765Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <div style=\"background-color: #7851A9; padding: 12px; font-size: 18px; color: #FFFFFF; border-left: 8px solid #C9A0DC\"><b>📊🔍 EDA</b>","metadata":{"papermill":{"duration":0.007425,"end_time":"2025-06-11T13:37:19.521105","exception":false,"start_time":"2025-06-11T13:37:19.513680","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## <div style=\"background-color: #7851A9; padding: 12px; font-size: 18px; color: #FFFFFF; border-left: 8px solid #C9A0DC\"><b>03 :🔬🧐 Data Inspection</b>\n\n<div style=\"background-color: #7851A9; border-left: 8px solid #C9A0DC; padding: 20px; border-radius: 8px; font-size: 14px; color: #FFFFFF;\">\n  <h5 style=\"font-size: 16px; margin-bottom: 10px; background-color: #4B0082; padding: 8px; border-radius: 4px; color: #FFFFFF;\">\n    <strong>Findings</strong> \n  </h5>\n    └ 1 : describe later<br>\n\n  <h5 style=\"font-size: 16px; margin-bottom: 10px; background-color: #4B0082; padding: 8px; border-radius: 4px; color: #FFFFFF;\">\n    <strong>Insights</strong> \n  </h5>\n    └ 1 : describe later<br>\n  </div>","metadata":{"papermill":{"duration":0.007398,"end_time":"2025-06-11T13:37:19.536065","exception":false,"start_time":"2025-06-11T13:37:19.528667","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Check if each column has a unique value of 0 and 1, and assign 1 or 0\ndef check_if_binary(column):\n    unique_values = column.unique()\n    if len(unique_values) == 2 and set(unique_values) == {0, 1}:\n        return 1\n    else:\n        return 0","metadata":{"papermill":{"duration":0.013397,"end_time":"2025-06-11T13:37:19.556915","exception":false,"start_time":"2025-06-11T13:37:19.543518","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:43:48.091562Z","iopub.execute_input":"2025-07-09T13:43:48.092039Z","iopub.status.idle":"2025-07-09T13:43:48.096097Z","shell.execute_reply.started":"2025-07-09T13:43:48.092018Z","shell.execute_reply":"2025-07-09T13:43:48.095409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def check_if_outliers(df):\n\n    outliers_rate_dict = {}\n\n    for column_name in df.columns:\n        column = df[column_name]\n\n        # Skip if not a numeric column\n        if not pd.api.types.is_numeric_dtype(column):\n            print(f\"'{column_name}' is skipped because it is not a numeric type.\")\n            continue\n\n        # Calculate the mean and standard deviation of the data\n        mean = column.mean()\n        std = column.std()\n\n        # Set outlier threshold\n        threshold = 2  # Adjust this value to change the outlier criteria\n\n        # Set conditions for detecting outliers\n        lower_bound = mean - threshold * std\n        upper_bound = mean + threshold * std\n\n        # Detect outliers\n        outliers = (column < lower_bound) | (column > upper_bound)\n\n        # Calculate the percentage of outliers\n        outliers_rate = outliers.sum() / len(column)\n\n        # Save results to dictionary\n        outliers_rate_dict[column_name] = outliers_rate\n\n    return outliers_rate_dict","metadata":{"papermill":{"duration":0.014551,"end_time":"2025-06-11T13:37:19.579031","exception":false,"start_time":"2025-06-11T13:37:19.564480","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:43:48.252105Z","iopub.execute_input":"2025-07-09T13:43:48.252601Z","iopub.status.idle":"2025-07-09T13:43:48.257356Z","shell.execute_reply.started":"2025-07-09T13:43:48.252578Z","shell.execute_reply":"2025-07-09T13:43:48.256777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def data_inspection(df):\n    \"\"\"A function that generates data inspection information for a data frame\"\"\"\n\n    # process_one:Create a DataFrame with basic statistics\n    data_inspection = pd.DataFrame({\n        'column_name'         : df.columns,\n        'data_type'           : df.dtypes,\n        'cnt_rows'            : len(df),\n        'cnt_unique_rows'     : df.nunique(),\n        'cnt_duplicated_rows' : len(df) - df.nunique(),\n        'cnt_non_null_rows'   : df.count().values,\n        'cnt_null_rows'       : df.isnull().sum(),\n        'rate_null_rows'      : (df.isnull().sum() / len(df)),\n        # 'date min': df['t_dat'].min(),\n        # 'date max': df['t_dat'].max()\n    })\n\n    # process_two:Calculate descriptive statistics, modes, and percentages of modes for numerical data:\n\n    # The below code is no longer needed as 'custom_description' provides the stats\n    # description = df.describe(include=np.number).T.reset_index().rename(columns={'index': 'column_name'})\n    # median = df.median(numeric_only=True).reset_index().rename(columns={'index': 'column_name', 0: 'median'})\n\n    # Calculate the most frequent value for each column\n    mode_values = {}\n    for column in df.columns:\n        try:\n            mode_values[column] = df[column].mode().iloc[0]\n        except IndexError:\n            mode_values[column] = None  # Handle cases with no mode\n\n    # Calculate the percentage of most frequent values\n    mode_rates = {}\n    for column in df.columns:\n        if column in mode_values and mode_values[column] is not None:\n            mode_value = mode_values[column]\n            mode_rate = (df[column] == mode_value).sum() / len(df)\n            mode_rates[column] = mode_rate\n        else:\n            mode_rates[column] = None\n\n    # Convert the most common value and its percentage into a data frame\n    df_mode = pd.DataFrame(list(mode_values.items()), columns=['column_name', 'mode'])\n    df_mode['rate mode'] = df_mode['column_name'].map(mode_rates)\n\n    # Custom statistics calculation without using df.describe for basic stats\n    def calculate_custom_stats(series):\n        q1 = series.quantile(0.25)\n        q3 = series.quantile(0.75)\n        iqr = q3 - q1\n        return pd.Series({\n            'count': series.count(),\n            'mean': series.mean(),\n            'std': series.std(),\n            # 'q1-1.5*QTR': q1 - 1.5 * iqr,\n            '0%': series.quantile(0),\n            '25%': q1,\n            '50%': series.median(),\n            '75%': q3,\n            '100%': series.quantile(1),\n            'lower_bound': series.mean() - 2 * series.std(),\n            'upper_bound': series.mean() + 2 * series.std()\n            # 'q3+1.5*QTR': q3 + 1.5 * iqr\n        })\n\n    df_numeric = df.select_dtypes(include=np.number)\n    custom_description = df_numeric.apply(calculate_custom_stats).T.reset_index(names='column_name')\n\n    # process_three:Combining DataFrames and adding outlier rates, skewness, and kurtosis:\n    data_inspection = pd.merge(data_inspection, custom_description, how='left', on='column_name')\n    data_inspection = pd.merge(data_inspection, df_mode, how='left', on='column_name')\n\n    # # The 'description' merge is no longer needed as 'custom_description' provides the stats\n    # data_inspection = pd.merge(data_inspection, description, how='left', on='column_name')\n\n    outliers_rate = check_if_outliers(df)\n    outliers_rate = pd.DataFrame(list(outliers_rate.items()), columns=['column_name', 'outliers_rate']).reset_index(drop=True)\n    data_inspection = pd.merge(data_inspection, outliers_rate, how='left', on='column_name')\n\n    # skewness and kurtosis\n    skew = df.select_dtypes(include=np.number).skew().reset_index().rename(columns={'index': 'column_name', 0: 'skewness'})\n    kurt = df.select_dtypes(include=np.number).kurt().reset_index().rename(columns={'index': 'column_name', 0: 'kurtosis'})\n\n    data_inspection = pd.merge(data_inspection, skew, how='left', on='column_name')\n    data_inspection = pd.merge(data_inspection, kurt, how='left', on='column_name')\n\n    # Calculating correlation coefficients (numeric columns only)\n    df_numeric = df.select_dtypes(include=np.number)\n    if not df_numeric.empty and df_numeric.shape[1] > 1:\n        try:\n            correlation_matrix = df_numeric.corr(numeric_only=True)\n\n            # Assuming the last column is the target variable for correlation\n            target_column = df_numeric.columns[-1]\n            if target_column in correlation_matrix.index:\n                target_corr = pd.DataFrame({'column_name': correlation_matrix.index, 'target correlation': correlation_matrix[target_column]})\n                data_inspection = pd.merge(data_inspection, target_corr, how='left', on='column_name')\n            else:\n                data_inspection['target correlation'] = np.nan\n        except Exception as e:\n            print(f\"Error calculating correlation: {e}\")\n            data_inspection['target correlation'] = np.nan\n    else:\n        data_inspection['target correlation'] = np.nan\n\n    # process_four:Adding more column information and example data\n    # --- MODIFIED PART START ---\n    data_examples = {}\n    for col in df.columns:\n        non_null_values = df[col].dropna()\n        if not non_null_values.empty:\n            # Randomly select a non-null value\n            example_value = non_null_values.sample(n=1).iloc[0]\n            data_examples[col] = str(example_value).replace('\\n', '<br>')\n        else:\n            data_examples[col] = None # Or an empty string, depending on preference\n\n    # process_four:Adding more column information and example data\n    data_inspection_else = pd.DataFrame({\n        'column_name': df.columns,\n        'flag_or_not': df.apply(check_if_binary),\n        'columns_details': None,\n        'remarks': None,\n        'trigger': None,\n        'dataset_name': None,\n        'existence_of_table_definition': None,\n        'data_exmaple': pd.Series(data_examples) # Assign the generated examples\n    })\n\n    data_inspection = pd.merge(data_inspection, data_inspection_else, how='left', on='column_name')\n\n    # visualization `data_inspection`\n    # blue → green → yellow\n    styled_columns = data_inspection.select_dtypes(include=np.number).columns\n    if not styled_columns.empty:\n        data_inspection_styled = data_inspection.style.background_gradient(cmap='viridis', subset=pd.IndexSlice[:, styled_columns])\n    else:\n        data_inspection_styled = data_inspection\n\n    return data_inspection_styled\n","metadata":{"papermill":{"duration":0.024529,"end_time":"2025-06-11T13:37:19.610959","exception":false,"start_time":"2025-06-11T13:37:19.586430","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:45:58.266176Z","iopub.execute_input":"2025-07-09T13:45:58.266919Z","iopub.status.idle":"2025-07-09T13:45:58.281492Z","shell.execute_reply.started":"2025-07-09T13:45:58.266897Z","shell.execute_reply":"2025-07-09T13:45:58.280867Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <div style=\"background-color: #7851A9; padding: 12px; font-size: 18px; color: #FFFFFF; border-left: 8px solid #C9A0DC\"><b>df_train</b>","metadata":{"papermill":{"duration":0.007452,"end_time":"2025-06-11T13:37:19.626167","exception":false,"start_time":"2025-06-11T13:37:19.618715","status":"completed"},"tags":[]}},{"cell_type":"code","source":"data_inspection_styled = data_inspection(df_articles)\ndisplay(data_inspection_styled)\n# data_inspection.to_csv('data_inspection.csv', index = 'false') # Saving the dataframe if necessary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:44:13.538907Z","iopub.execute_input":"2025-07-09T13:44:13.539478Z","iopub.status.idle":"2025-07-09T13:44:14.634234Z","shell.execute_reply.started":"2025-07-09T13:44:13.539458Z","shell.execute_reply":"2025-07-09T13:44:14.633502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_inspection_styled = data_inspection(df_customers)\ndisplay(data_inspection_styled)\n# data_inspection.to_csv('data_inspection.csv', index = 'false') # Saving the dataframe if necessary","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:44:14.747126Z","iopub.execute_input":"2025-07-09T13:44:14.747389Z","iopub.status.idle":"2025-07-09T13:44:23.066445Z","shell.execute_reply.started":"2025-07-09T13:44:14.747371Z","shell.execute_reply":"2025-07-09T13:44:23.065795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_inspection_styled = data_inspection(df_train)\ndisplay(data_inspection_styled)\n# data_inspection.to_csv('data_inspection.csv', index = 'false') # Saving the dataframe if necessary","metadata":{"papermill":{"duration":2.269356,"end_time":"2025-06-11T13:37:21.959848","exception":false,"start_time":"2025-06-11T13:37:19.690492","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:46:02.741119Z","iopub.execute_input":"2025-07-09T13:46:02.741384Z","iopub.status.idle":"2025-07-09T13:47:19.208493Z","shell.execute_reply.started":"2025-07-09T13:46:02.741364Z","shell.execute_reply":"2025-07-09T13:47:19.207771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(len(df_articles) == len(df_articles.drop_duplicates()))\ndisplay(len(df_customers) == len(df_customers.drop_duplicates()))\ndisplay(len(df_train) == len(df_train.drop_duplicates()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T13:57:04.828734Z","iopub.execute_input":"2025-07-09T13:57:04.829504Z","iopub.status.idle":"2025-07-09T13:57:23.645059Z","shell.execute_reply.started":"2025-07-09T13:57:04.829474Z","shell.execute_reply":"2025-07-09T13:57:23.644423Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## df_articles","metadata":{}},{"cell_type":"code","source":"df_articles.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:04:41.087652Z","iopub.execute_input":"2025-07-09T14:04:41.087932Z","iopub.status.idle":"2025-07-09T14:04:41.100946Z","shell.execute_reply.started":"2025-07-09T14:04:41.087913Z","shell.execute_reply":"2025-07-09T14:04:41.100404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of columns to visualize\ncolumns_to_visualize = [\n    'prod_name',\n    'product_type_name',\n    'product_group_name',\n    'department_name',\n    'index_name',\n    'index_group_name',\n    'section_name',\n    'garment_group_name',\n    'detail_desc'\n]\n\n# Loop through each column and create a bar plot\nfor col_name in columns_to_visualize:\n    # Get the top 50 values for the current column\n    # value_counts()の実行前にカテゴリカル型に変換することで、\n    # 欠損値（NaN）がある場合にエラーが発生するのを防ぎ、\n    # 欠損値をカウントから除外できます。\n    top_50_values = df_articles[col_name].astype(str).value_counts().head(50)\n\n    # Create a new figure for each plot\n    plt.figure(figsize=(12, max(6, len(top_50_values) * 0.4))) # Adjust height based on number of bars\n\n    # Create the bar plot\n    sns.barplot(x=top_50_values.values, y=top_50_values.index, palette='viridis')\n\n    # Set title and labels dynamically\n    plt.title(f'Top 50 {col_name.replace(\"_\", \" \").title()} by Count', fontsize=16)\n    plt.xlabel('Count', fontsize=12)\n    plt.ylabel(col_name.replace(\"_\", \" \").title(), fontsize=12)\n\n    # Adjust layout\n    plt.tight_layout()\n\n    # Display the plot\n    plt.show()\n\nprint(\"All plots displayed successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:11:43.893817Z","iopub.execute_input":"2025-07-09T14:11:43.894499Z","iopub.status.idle":"2025-07-09T14:11:48.082232Z","shell.execute_reply.started":"2025-07-09T14:11:43.894476Z","shell.execute_reply":"2025-07-09T14:11:48.081530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## df_customers","metadata":{}},{"cell_type":"code","source":"df_customers.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:02:27.629782Z","iopub.execute_input":"2025-07-09T14:02:27.630321Z","iopub.status.idle":"2025-07-09T14:02:27.640185Z","shell.execute_reply.started":"2025-07-09T14:02:27.630300Z","shell.execute_reply":"2025-07-09T14:02:27.639431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of columns to visualize\ncolumns_to_visualize = [\n    'FN',\n    'Active',\n    'club_member_status',\n    'fashion_news_frequency',\n    'age'\n]\n\n# Loop through each column and create a bar plot\nfor col_name in columns_to_visualize:\n    # Get the top 50 values for the current column\n    # value_counts()の実行前にカテゴリカル型に変換することで、\n    # 欠損値（NaN）がある場合にエラーが発生するのを防ぎ、\n    # 欠損値をカウントから除外できます。\n    top_50_values = df_customers[col_name].astype(str).value_counts().head(50)\n\n    # Create a new figure for each plot\n    plt.figure(figsize=(12, max(6, len(top_50_values) * 0.4))) # Adjust height based on number of bars\n\n    # Create the bar plot\n    sns.barplot(x=top_50_values.values, y=top_50_values.index, palette='viridis')\n\n    # Set title and labels dynamically\n    plt.title(f'Top 50 {col_name.replace(\"_\", \" \").title()} by Count', fontsize=16)\n    plt.xlabel('Count', fontsize=12)\n    plt.ylabel(col_name.replace(\"_\", \" \").title(), fontsize=12)\n\n    # Adjust layout\n    plt.tight_layout()\n\n    # Display the plot\n    plt.show()\n\nprint(\"All plots displayed successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:14:14.247544Z","iopub.execute_input":"2025-07-09T14:14:14.248291Z","iopub.status.idle":"2025-07-09T14:14:17.851073Z","shell.execute_reply.started":"2025-07-09T14:14:14.248270Z","shell.execute_reply":"2025-07-09T14:14:17.850395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## df_train","metadata":{}},{"cell_type":"code","source":"display(len(df_train))\ndf_train['t_dat'] = pd.to_datetime(df_train['t_dat'], errors='coerce')\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:21:38.684049Z","iopub.execute_input":"2025-07-09T14:21:38.684744Z","iopub.status.idle":"2025-07-09T14:21:41.774073Z","shell.execute_reply.started":"2025-07-09T14:21:38.684719Z","shell.execute_reply":"2025-07-09T14:21:41.773457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top_50_detail_desc = df_train['t_dat'].value_counts().head(50)\n\nplt.figure(figsize=(12, 10))\nsns.barplot(x=top_50_detail_desc.values, y=top_50_detail_desc.index, palette='viridis')\n\nplt.title('Top 50 t_dat by Count', fontsize=16)\nplt.xlabel('Count', fontsize=12)\nplt.ylabel('t_dat', fontsize=12)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:23:25.566639Z","iopub.execute_input":"2025-07-09T14:23:25.567100Z","iopub.status.idle":"2025-07-09T14:23:26.274102Z","shell.execute_reply.started":"2025-07-09T14:23:25.567077Z","shell.execute_reply":"2025-07-09T14:23:26.273400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['t_dat'].max()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:24:26.283061Z","iopub.execute_input":"2025-07-09T14:24:26.283587Z","iopub.status.idle":"2025-07-09T14:24:26.364491Z","shell.execute_reply.started":"2025-07-09T14:24:26.283562Z","shell.execute_reply":"2025-07-09T14:24:26.363889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 最新の日付を取得\nmax_date = df_train['t_dat'].max()\nthree_months_ago = max_date - pd.DateOffset(months=3)\ndf_last_3_months = df_train[df_train['t_dat'] >= three_months_ago]\n\ndisplay(three_months_ago), display(max_date)\ndf_last_3_months","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:27:37.071178Z","iopub.execute_input":"2025-07-09T14:27:37.071974Z","iopub.status.idle":"2025-07-09T14:27:37.456838Z","shell.execute_reply.started":"2025-07-09T14:27:37.071941Z","shell.execute_reply":"2025-07-09T14:27:37.456195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top_50_detail_desc = df_last_3_months['t_dat'].value_counts().head(50)\n\nplt.figure(figsize=(12, 10))\nsns.barplot(x=top_50_detail_desc.values, y=top_50_detail_desc.index, palette='viridis')\n\nplt.title('Top 50 t_dat by Count', fontsize=16)\nplt.xlabel('Count', fontsize=12)\nplt.ylabel('t_dat', fontsize=12)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:26:56.516015Z","iopub.execute_input":"2025-07-09T14:26:56.516307Z","iopub.status.idle":"2025-07-09T14:26:57.211550Z","shell.execute_reply.started":"2025-07-09T14:26:56.516286Z","shell.execute_reply":"2025-07-09T14:26:57.210767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_merge_trainsaction_customer = pd.merge(df_last_3_months, df_customers, how = 'inner', on  = 'customer_id')\ndf_merge_trainsaction_customer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:29:07.381119Z","iopub.execute_input":"2025-07-09T14:29:07.381606Z","iopub.status.idle":"2025-07-09T14:29:09.343901Z","shell.execute_reply.started":"2025-07-09T14:29:07.381583Z","shell.execute_reply":"2025-07-09T14:29:09.343155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of columns to visualize\ncolumns_to_visualize = [\n    'FN',\n    'Active',\n    'club_member_status',\n    'fashion_news_frequency',\n    'age'\n]\n\n# Loop through each column and create a bar plot\nfor col_name in columns_to_visualize:\n    # Get the top 50 values for the current column\n    # value_counts()の実行前にカテゴリカル型に変換することで、\n    # 欠損値（NaN）がある場合にエラーが発生するのを防ぎ、\n    # 欠損値をカウントから除外できます。\n    top_50_values = df_merge_trainsaction_customer[col_name].astype(str).value_counts().head(50)\n\n    # Create a new figure for each plot\n    plt.figure(figsize=(12, max(6, len(top_50_values) * 0.4))) # Adjust height based on number of bars\n\n    # Create the bar plot\n    sns.barplot(x=top_50_values.values, y=top_50_values.index, palette='viridis')\n\n    # Set title and labels dynamically\n    plt.title(f'Top 50 {col_name.replace(\"_\", \" \").title()} by Count', fontsize=16)\n    plt.xlabel('Count', fontsize=12)\n    plt.ylabel(col_name.replace(\"_\", \" \").title(), fontsize=12)\n\n    # Adjust layout\n    plt.tight_layout()\n\n    # Display the plot\n    plt.show()\n\nprint(\"All plots displayed successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:29:31.933690Z","iopub.execute_input":"2025-07-09T14:29:31.934398Z","iopub.status.idle":"2025-07-09T14:29:40.047970Z","shell.execute_reply.started":"2025-07-09T14:29:31.934373Z","shell.execute_reply":"2025-07-09T14:29:40.047225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_merge_trainsaction_article = pd.merge(df_last_3_months, df_articles, how = 'inner', on  = 'article_id')\ndf_merge_trainsaction_article","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:30:47.321975Z","iopub.execute_input":"2025-07-09T14:30:47.322222Z","iopub.status.idle":"2025-07-09T14:30:49.055858Z","shell.execute_reply.started":"2025-07-09T14:30:47.322205Z","shell.execute_reply":"2025-07-09T14:30:49.055109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of columns to visualize\ncolumns_to_visualize = [\n    'prod_name',\n    'product_type_name',\n    'product_group_name',\n    'department_name',\n    'index_name',\n    'index_group_name',\n    'section_name',\n    'garment_group_name',\n    'detail_desc'\n]\n\n# Loop through each column and create a bar plot\nfor col_name in columns_to_visualize:\n    # Get the top 50 values for the current column\n    # value_counts()の実行前にカテゴリカル型に変換することで、\n    # 欠損値（NaN）がある場合にエラーが発生するのを防ぎ、\n    # 欠損値をカウントから除外できます。\n    top_50_values = df_merge_trainsaction_article[col_name].astype(str).value_counts().head(50)\n\n    # Create a new figure for each plot\n    plt.figure(figsize=(12, max(6, len(top_50_values) * 0.4))) # Adjust height based on number of bars\n\n    # Create the bar plot\n    sns.barplot(x=top_50_values.values, y=top_50_values.index, palette='viridis')\n\n    # Set title and labels dynamically\n    plt.title(f'Top 50 {col_name.replace(\"_\", \" \").title()} by Count', fontsize=16)\n    plt.xlabel('Count', fontsize=12)\n    plt.ylabel(col_name.replace(\"_\", \" \").title(), fontsize=12)\n\n    # Adjust layout\n    plt.tight_layout()\n\n    # Display the plot\n    plt.show()\n\nprint(\"All plots displayed successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T14:31:35.771218Z","iopub.execute_input":"2025-07-09T14:31:35.771865Z","iopub.status.idle":"2025-07-09T14:31:42.534658Z","shell.execute_reply.started":"2025-07-09T14:31:35.771840Z","shell.execute_reply":"2025-07-09T14:31:42.533923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}