{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Step 1: Imports and Reading Data","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pylab as plt #for visualizing dataset\nimport seaborn as sns\n#for using style in matplotlib and seaborn style need to inslall these packages\nplt.style.use('ggplot')","metadata":{"execution":{"iopub.status.busy":"2023-10-16T04:45:31.595228Z","iopub.execute_input":"2023-10-16T04:45:31.596242Z","iopub.status.idle":"2023-10-16T04:45:33.068621Z","shell.execute_reply.started":"2023-10-16T04:45:31.596204Z","shell.execute_reply":"2023-10-16T04:45:33.067491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Read the dataset\n\ndf = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-16T06:09:02.498006Z","iopub.execute_input":"2023-10-16T06:09:02.498374Z","iopub.status.idle":"2023-10-16T06:09:03.223043Z","shell.execute_reply.started":"2023-10-16T06:09:02.498344Z","shell.execute_reply":"2023-10-16T06:09:03.221778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customers = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\ndf_transaction_train = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-16T04:45:34.302228Z","iopub.execute_input":"2023-10-16T04:45:34.302691Z","iopub.status.idle":"2023-10-16T04:46:58.803302Z","shell.execute_reply.started":"2023-10-16T04:45:34.302662Z","shell.execute_reply":"2023-10-16T04:46:58.802392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_transaction_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-13T09:03:22.601212Z","iopub.execute_input":"2023-10-13T09:03:22.601676Z","iopub.status.idle":"2023-10-13T09:03:22.627385Z","shell.execute_reply.started":"2023-10-13T09:03:22.601642Z","shell.execute_reply":"2023-10-13T09:03:22.626440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customers.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:29.862635Z","iopub.execute_input":"2023-10-12T07:53:29.863730Z","iopub.status.idle":"2023-10-12T07:53:29.879105Z","shell.execute_reply.started":"2023-10-12T07:53:29.863684Z","shell.execute_reply":"2023-10-12T07:53:29.877871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:29.880876Z","iopub.execute_input":"2023-10-12T07:53:29.881332Z","iopub.status.idle":"2023-10-12T07:53:29.912193Z","shell.execute_reply.started":"2023-10-12T07:53:29.881303Z","shell.execute_reply":"2023-10-12T07:53:29.910663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 2: Data Understanding\nWhat I can do here:\n* Shape of the Dataframe\n* Head and tail of Dataframe\n* Data types\n* Description of Dataset","metadata":{}},{"cell_type":"code","source":"#Shape of Dataframe\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:29.914169Z","iopub.execute_input":"2023-10-12T07:53:29.914627Z","iopub.status.idle":"2023-10-12T07:53:29.928552Z","shell.execute_reply.started":"2023-10-12T07:53:29.914597Z","shell.execute_reply":"2023-10-12T07:53:29.927509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset has 105542 rows and 25 Columns","metadata":{}},{"cell_type":"code","source":"#To see the first couple of rows of the dataset\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:29.929852Z","iopub.execute_input":"2023-10-12T07:53:29.930267Z","iopub.status.idle":"2023-10-12T07:53:29.960104Z","shell.execute_reply.started":"2023-10-12T07:53:29.930227Z","shell.execute_reply":"2023-10-12T07:53:29.958666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* By default, it shows us the first 5 columns of the dataset, but if we want to see more, we can put the desired number in the bracket","metadata":{}},{"cell_type":"code","source":"#Due to the higher number of columns in the dataset, it does not show us all the columns in the \"HEAD\" function. To see all of them we can load this function:\npd.set_option('display.max_columns', 200)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:29.962182Z","iopub.execute_input":"2023-10-12T07:53:29.962663Z","iopub.status.idle":"2023-10-12T07:53:29.985604Z","shell.execute_reply.started":"2023-10-12T07:53:29.962622Z","shell.execute_reply":"2023-10-12T07:53:29.984343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#To see all the columns name\ndf.columns","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:29.988460Z","iopub.execute_input":"2023-10-12T07:53:29.988982Z","iopub.status.idle":"2023-10-12T07:53:30.001718Z","shell.execute_reply.started":"2023-10-12T07:53:29.988833Z","shell.execute_reply":"2023-10-12T07:53:30.000248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Articles\n* **article_id** : A unique identifier of every article.\n* **product_code, prod_name** : A unique identifier of every product and its name\n* **product_type, product_type_name** : The group of product_code and its name\n* **graphical_appearance_no, graphical_appearance_name** : The group of graphics and its name\n* **colour_group_code, colour_group_name** : The group of color and its name\n* **perceived_colour_value_id, perceived_colour_value_name, perceived_colour_master_id, perceived_colour_master_name** : The added color info\n* **department_no, department_name**: A unique identifier of every dep and its name\n* **index_code, index_name**: A unique identifier of every index and its name\n* **index_group_no, index_group_name**: A group of indices and its name\n* **section_no, section_name**: A unique identifier of every section and its name\n* **garment_group_no, garment_group_name**: A unique identifier of every garment and its name\n* **detail_desc**: Details","metadata":{}},{"cell_type":"markdown","source":"# Transaction\n\n1. **t_dat:** This column represents the date of the transaction. It appears to be in a YYYY-MM-DD format, indicating when each sale occurred.\n\n2. **customer_id:** This column contains unique identifiers for customers. Each customer is associated with a distinct ID, which can be used to track their purchases and behavior.\n\n3. **article_id:** This column contains identifiers for the articles or products sold. Each product has a unique ID, allowing you to identify which specific items were purchased.\n\n4. **price:** This column represents the price of the article or product in the local currency (e.g., dollars, euros, etc.). Prices are expressed as numerical values.\n\n5. **sales_channel_id:** This column contains identifiers for sales channels or channels through which the products were sold. Each channel is associated with a distinct ID, indicating where the sales took place (e.g., physical stores, online platforms, etc.).\n\n.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objs as go\nfrom plotly.subplots import make_subplots\n\n# Sample DataFrame (replace with your actual data)\n# Assuming you have a DataFrame df_transaction_train with 't_dat' and 'price' columns\n# You can use your actual data like this:\n# df_transaction_train = pd.read_csv('your_data.csv')  # Replace with your data source\n\n# Create a figure with subplots\nfig = make_subplots(rows=1, cols=1)\n\n# Create a time series plot\ntrace = go.Scatter(x=df_transaction_train['t_dat'], y=df_transaction_train['price'], mode='lines+markers', name='Price')\nfig.add_trace(trace)\n\n# Get the minimum and maximum dates from the transaction table\nmin_date = df_transaction_train['t_dat'].min()\nmax_date = df_transaction_train['t_dat'].max()\n\n# Define step options for the slider\nstep_options = [\n    dict(method='relayout', args=['xaxis', {'range': [min_date, max_date]}], label='All'),\n    dict(method='relayout', args=['xaxis', {'range': [max_date - pd.DateOffset(years=1), max_date]}], label='Yearly'),\n    dict(method='relayout', args=['xaxis', {'range': [max_date - pd.DateOffset(months=1), max_date]}], label='Monthly'),\n    dict(method='relayout', args=['xaxis', {'range': [max_date - pd.DateOffset(weeks=1), max_date]}], label='Weekly'),\n    dict(method='relayout', args=['xaxis', {'range': [max_date - pd.DateOffset(days=1), max_date]}], label='Daily')\n]\n\n# Create a slider for adjusting the date range\ndate_slider = dict(\n    active=2,  # Index of the default step (Monthly)\n    currentvalue=dict(prefix=\"Time Interval: \"),\n    steps=[dict(label=step['label'], method=step['method'], args=step['args']) for step in step_options]\n)\n\n# Update layout to include the slider\nfig.update_layout(\n    sliders=[date_slider],\n    title=\"Sales Over Time with Date Range Slider\",\n    xaxis_title=\"Date\",\n    yaxis_title=\"Sales\"\n)\n\n# Show the interactive plot\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:30.007899Z","iopub.execute_input":"2023-10-12T07:53:30.008335Z","iopub.status.idle":"2023-10-12T07:53:51.900671Z","shell.execute_reply.started":"2023-10-12T07:53:30.008296Z","shell.execute_reply":"2023-10-12T07:53:51.899484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ARTICLE EDA","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.901629Z","iopub.status.idle":"2023-10-12T07:53:51.902370Z","shell.execute_reply.started":"2023-10-12T07:53:51.902199Z","shell.execute_reply":"2023-10-12T07:53:51.902216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":":\n\n1. **article_id:** \n   - The `article_id` is a unique identifier for each article.\n   - The minimum value is 108,775, and the maximum value is 959,461.\n\n2. **product_code:**\n   - The `product_code` is another unique identifier for each product.\n   - The minimum value is 108,775, and the maximum value is 959,461.\n\n3. **product_type_no:**\n   - The `product_type_no` represents the product type, with a mean value of approximately 234.86.\n   - The minimum value is -1, which might indicate missing or undefined values.\n   - The maximum value is 762.\n\n4. **graphical_appearance_no:**\n   - The `graphical_appearance_no` represents the appearance of graphics associated with the articles.\n   - The mean value is around 1,009,515, and the minimum value is -1, which might indicate missing or undefined values.\n\n5. **colour_group_code:**\n   - The `colour_group_code` represents the code for color groups.\n   - The mean value is approximately 32.23, with a minimum value of -1 (possibly missing/undefined) and a maximum of 93.\n\n6. **perceived_colour_value_id:**\n   - This column seems to represent perceived color values.\n   - The mean value is approximately 3.21, with a minimum value of -1 (possibly missing/undefined) and a maximum of 7.\n\n7. **perceived_colour_master_id:**\n   - This column also relates to perceived color values.\n   - The mean value is around 7.81, with a minimum value of -1 (possibly missing/undefined) and a maximum of 20.\n\n8. **department_no:**\n   - The `department_no` is associated with different departments.\n   - The mean value is approximately 4532.78, with a minimum of 1201 and a maximum of 9989.\n\n9. **index_group_no:**\n   - This column represents the index group number.\n   - The mean value is around 3.17, with a minimum value of 1 and a maximum of 26.\n\n10. **section_no:**\n    - The `section_no` is related to different sections.\n    - The mean value is approximately 42.66, with a minimum of 2 and a maximum of 97.\n\n11. **garment_group_no:**\n    - The `garment_group_no` is associated with garment groups.\n    - The mean value is around 1010.44, with a minimum value of 1001 and a maximum of 1025.\n\nThis summary provides an overview of the distribution of various columns in the `df` ","metadata":{}},{"cell_type":"code","source":"df['article_id'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.903818Z","iopub.status.idle":"2023-10-12T07:53:51.904322Z","shell.execute_reply.started":"2023-10-12T07:53:51.904161Z","shell.execute_reply":"2023-10-12T07:53:51.904178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['product_code'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.905164Z","iopub.status.idle":"2023-10-12T07:53:51.906256Z","shell.execute_reply.started":"2023-10-12T07:53:51.906047Z","shell.execute_reply":"2023-10-12T07:53:51.906066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"z = df.groupby('product_code')['article_id'].count().reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.907713Z","iopub.status.idle":"2023-10-12T07:53:51.908318Z","shell.execute_reply.started":"2023-10-12T07:53:51.908091Z","shell.execute_reply":"2023-10-12T07:53:51.908115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['product_code']==108775]","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.909497Z","iopub.status.idle":"2023-10-12T07:53:51.909886Z","shell.execute_reply.started":"2023-10-12T07:53:51.909686Z","shell.execute_reply":"2023-10-12T07:53:51.909704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# making unique product combination","metadata":{}},{"cell_type":"code","source":"# Concatenate the relevant columns to create a unique product identifier\ndf['unique_product'] = df.apply(\n    lambda row: (\n        row['product_code'],\n        row['product_type_no'],\n        row['graphical_appearance_no'],\n        row['colour_group_code'],\n        row['perceived_colour_value_id'],\n        row['perceived_colour_master_id']\n    ),\n    axis=1\n)\n\n# Find the total number of unique products\ntotal_unique_products = df['unique_product'].nunique()\n\n# List unique combinations of product attributes\nunique_product_combinations = df['unique_product'].unique()\n\n# Display the results\nprint(\"Total Unique Products:\", total_unique_products)\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.911442Z","iopub.status.idle":"2023-10-12T07:53:51.912073Z","shell.execute_reply.started":"2023-10-12T07:53:51.911884Z","shell.execute_reply":"2023-10-12T07:53:51.911904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Total Unique Products: 98872","metadata":{}},{"cell_type":"markdown","source":"1. Now check how much sales they have made","metadata":{}},{"cell_type":"code","source":"# Merge the sales data with the product data based on 'article_id'\nmerged_data = df.merge(df_transaction_train, on='article_id', how='inner')\n\n# Calculate total sales per unique product\ntotal_sales_per_product = merged_data.groupby('unique_product')['price'].sum()\n\n# Display the total sales for each unique product\nprint(\"Total Sales for Unique Products:\")\nprint(total_sales_per_product)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.912963Z","iopub.status.idle":"2023-10-12T07:53:51.913279Z","shell.execute_reply.started":"2023-10-12T07:53:51.913128Z","shell.execute_reply":"2023-10-12T07:53:51.913142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TSPP = pd.DataFrame(total_sales_per_product).reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.915106Z","iopub.status.idle":"2023-10-12T07:53:51.915721Z","shell.execute_reply.started":"2023-10-12T07:53:51.915518Z","shell.execute_reply":"2023-10-12T07:53:51.915542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TSPP.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.917444Z","iopub.status.idle":"2023-10-12T07:53:51.917845Z","shell.execute_reply.started":"2023-10-12T07:53:51.917653Z","shell.execute_reply":"2023-10-12T07:53:51.917673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\n# Assuming you have the total sales per unique product in a DataFrame\n# Replace 'total_sales_per_product_df' with the actual DataFrame containing your data\n\n# Sort the DataFrame by total sales in descending order and select the top 10 products\ntop_10_products = TSPP.sort_values('price', ascending=False).head(10)\nlast_10_products = TSPP.sort_values('price', ascending=True).head(10)\n\ntop_10_products\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.919126Z","iopub.status.idle":"2023-10-12T07:53:51.919551Z","shell.execute_reply.started":"2023-10-12T07:53:51.919330Z","shell.execute_reply":"2023-10-12T07:53:51.919350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_10_products","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.921341Z","iopub.status.idle":"2023-10-12T07:53:51.921742Z","shell.execute_reply.started":"2023-10-12T07:53:51.921557Z","shell.execute_reply":"2023-10-12T07:53:51.921578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"(706016, 272, 1010016, 9, 4, 5)","metadata":{}},{"cell_type":"code","source":"df[df['unique_product']==(344625, 76, 1010017, 71, 1, 2)]","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.922937Z","iopub.status.idle":"2023-10-12T07:53:51.923284Z","shell.execute_reply.started":"2023-10-12T07:53:51.923116Z","shell.execute_reply":"2023-10-12T07:53:51.923132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Findings\n1. Jade HW Skinny Denim TRS top unique sellig product in H&M  total_sale of-->1786.561831\n\n\n2. ELLEN Sunhat is the lowest unique selling product in H&M total_sale of--> 0.000424\n","metadata":{}},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.925123Z","iopub.status.idle":"2023-10-12T07:53:51.925507Z","shell.execute_reply.started":"2023-10-12T07:53:51.925294Z","shell.execute_reply":"2023-10-12T07:53:51.925310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport plotly.express as px\nimport matplotlib.pyplot as plt\n\n\n\n# 1. Pie Chart for Product Types (using Plotly)\nfig = px.pie(df, names='product_type_name', title='Distribution of Product Types', color_discrete_sequence=px.colors.qualitative.Set3)\n\nfig.show()\n\n# 2. Bar Chart for Colour Groups (using Plotly)\ncolour_group_counts = df['colour_group_name'].value_counts()\nfig = px.bar(colour_group_counts, x=colour_group_counts.index, y=colour_group_counts.values, labels={'x': 'Colour Group', 'y': 'Count'}, title='Colour Group Distribution', color_discrete_sequence=px.colors.qualitative.Pastel)\nfig.show()\n\n# 3. Scatter Plot for Colour and Department (using Matplotlib)\nplt.figure(figsize=(10, 6))\nplt.scatter(df['colour_group_code'], df['department_no'], alpha=0.5, c='green')\nplt.xlabel('Colour Group Code')\nplt.ylabel('Department Number')\nplt.title('Scatter Plot of Colour Group vs. Department')\nplt.show()\n\n# 4. Interactive Heatmap (using Plotly)\nfig = px.imshow(df.corr(), title='Correlation Heatmap', color_continuous_scale='Viridis')\nfig.show()\n\n# 5. Interactive Histogram for Article IDs (using Plotly)\nfig = px.histogram(df, x='article_id', title='Distribution of Article IDs', color_discrete_sequence=['blue'])\nfig.show()\n\n# 6. Interactive 3D Scatter Plot (using Plotly)\nfig = px.scatter_3d(df, x='product_type_no', y='colour_group_code', z='department_no', title='3D Scatter Plot', color='perceived_colour_value_id')\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.926973Z","iopub.status.idle":"2023-10-12T07:53:51.928476Z","shell.execute_reply.started":"2023-10-12T07:53:51.927966Z","shell.execute_reply":"2023-10-12T07:53:51.928013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\nfig = px.scatter_3d(\n    df,\n    x='product_type_no',\n    y='colour_group_code',\n    z='perceived_colour_value_id',\n    color='department_name',  # Color by department_name\n    size='garment_group_no'  # Size by garment_group_no\n)\n\n# Customize the appearance\nfig.update_layout(\n    scene=dict(\n        xaxis_title='Product Type No',\n        yaxis_title='Colour Group Code',\n        zaxis_title='Perceived Colour Value ID'\n    )\n)\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.930203Z","iopub.status.idle":"2023-10-12T07:53:51.931166Z","shell.execute_reply.started":"2023-10-12T07:53:51.930959Z","shell.execute_reply":"2023-10-12T07:53:51.930981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The 3D scatter plot can provide various insights depending on the context and the actual data. Here are some potential insights you can gain from the graph I provided:\n\n1. **Distribution of Products by Type, Color, and Perceived Color Value:**\n   - You can observe the distribution of products in a 3D space, where the X-axis represents the product type, the Y-axis represents the color group code, and the Z-axis represents the perceived color value. This allows you to see how products are spread across these three dimensions.\n\n2. **Color Department Representation:**\n   - The color of data points in the graph is based on the 'department_name'. You can identify which departments are associated with different regions of the 3D space. For example, if certain departments cluster in a particular area of the plot, it suggests that those departments share similar product characteristics in terms of product type, color, and perceived color value.\n\n3. **Size Significance:**\n   - The size of data points is determined by 'garment_group_no'. Larger markers may indicate higher values in this category, and you can assess if there's any pattern related to the size of the garments within the 3D space.\n\n4. **Outliers and Clusters:**\n   - Identify any outliers or clusters within the 3D space. Outliers may represent products that are unique or don't fit the typical patterns observed in the dataset. Clusters can indicate groups of products that share similar attributes.\n\n5. **Correlations and Relationships:**\n   - By examining the distribution of data points, you can explore potential correlations and relationships between the three chosen dimensions. For example, you might notice that certain color groups are associated with specific product types and perceived color values.\n\n6. **Data Quality:**\n   - If you notice data points concentrated at certain extreme values or near -1 (which might indicate missing or undefined values), it could indicate data quality issues in those dimensions.\n\n","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\n\nfig = px.scatter_3d(\n    df,\n    x='product_type_no',\n    y='colour_group_code',\n    z='perceived_colour_value_id',\n    color='department_name',  # Color by department_name\n    size='garment_group_no'  # Size by garment_group_no\n)\n\n# Customize the appearance\nfig.update_layout(\n    scene=dict(\n        xaxis_title='Product Type No',\n        yaxis_title='Colour Group Code',\n        zaxis_title='Perceived Colour Value ID'\n    )\n)\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.932394Z","iopub.status.idle":"2023-10-12T07:53:51.932803Z","shell.execute_reply.started":"2023-10-12T07:53:51.932629Z","shell.execute_reply":"2023-10-12T07:53:51.932646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter_3d(\n    df,\n    x='colour_group_code',\n    y='perceived_colour_master_id',\n    z='department_no',\n    color='department_name',\n    size='garment_group_no'\n)\nfig.update_layout(\n    scene=dict(\n        xaxis_title='Colour Group Code',\n        yaxis_title='Perceived Colour Master ID',\n        zaxis_title='Department No'\n    )\n)\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-12T07:53:51.935023Z","iopub.status.idle":"2023-10-12T07:53:51.935699Z","shell.execute_reply.started":"2023-10-12T07:53:51.935517Z","shell.execute_reply":"2023-10-12T07:53:51.935535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import statsmodels.api as sm\nfrom statsmodels.tsa.stattools import kpss","metadata":{"execution":{"iopub.status.busy":"2023-10-13T09:04:04.567608Z","iopub.execute_input":"2023-10-13T09:04:04.568053Z","iopub.status.idle":"2023-10-13T09:04:05.250743Z","shell.execute_reply.started":"2023-10-13T09:04:04.568022Z","shell.execute_reply":"2023-10-13T09:04:05.249333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# from KPSS checking if data is stationary or not\n\n1. H0 ---> data is stationary\n2. HA-----> data is not stationary","metadata":{}},{"cell_type":"code","source":"kpss(df_transaction_train['price'],'ct')","metadata":{"execution":{"iopub.status.busy":"2023-10-13T08:32:01.959428Z","iopub.execute_input":"2023-10-13T08:32:01.960494Z","iopub.status.idle":"2023-10-13T08:32:46.815060Z","shell.execute_reply.started":"2023-10-13T08:32:01.960415Z","shell.execute_reply":"2023-10-13T08:32:46.813365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# as  we can see the p_values is <-0.5 hance we can reject the null  so the data is non-stationary","metadata":{}},{"cell_type":"code","source":"df_transaction_train['t_dat'] = pd.to_datetime(df_transaction_train['t_dat'])\n","metadata":{"execution":{"iopub.status.busy":"2023-10-13T09:08:10.184674Z","iopub.execute_input":"2023-10-13T09:08:10.185163Z","iopub.status.idle":"2023-10-13T09:08:14.804289Z","shell.execute_reply.started":"2023-10-13T09:08:10.185120Z","shell.execute_reply":"2023-10-13T09:08:14.803019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_transaction_train = df_transaction_train.set_index('t_dat')","metadata":{"execution":{"iopub.status.busy":"2023-10-13T08:36:31.633616Z","iopub.execute_input":"2023-10-13T08:36:31.634037Z","iopub.status.idle":"2023-10-13T08:36:32.991736Z","shell.execute_reply.started":"2023-10-13T08:36:31.634006Z","shell.execute_reply":"2023-10-13T08:36:32.990534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Time Series Decomposition","metadata":{}},{"cell_type":"code","source":"df_transaction_train","metadata":{"execution":{"iopub.status.busy":"2023-10-13T09:08:33.624321Z","iopub.execute_input":"2023-10-13T09:08:33.624780Z","iopub.status.idle":"2023-10-13T09:08:33.644643Z","shell.execute_reply.started":"2023-10-13T09:08:33.624747Z","shell.execute_reply":"2023-10-13T09:08:33.642942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the plot\nplt.figure(figsize=(10, 6))\nplt.plot(df_transaction_train['t_dat'], df_transaction_train['price'], marker='o', linestyle='-')\nplt.xlabel('Date')\nplt.ylabel('Price')\nplt.title('Price Over Time')\nplt.grid(True)\nplt.show()\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-13T09:10:51.183816Z","iopub.execute_input":"2023-10-13T09:10:51.184882Z","iopub.status.idle":"2023-10-13T09:12:01.618855Z","shell.execute_reply.started":"2023-10-13T09:10:51.184817Z","shell.execute_reply":"2023-10-13T09:12:01.617597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objs as go\nfrom plotly.subplots import make_subplots\n\n# Sample DataFrame (replace with your actual data)\n# Assuming you have a DataFrame df_transaction_train with 't_dat' and 'price' columns\n# You can use your actual data like this:\n# df_transaction_train = pd.read_csv('your_data.csv')  # Replace with your data source\n\n# Create a figure with subplots\nfig = make_subplots(rows=1, cols=1)\n\n# Create a time series plot\ntrace = go.Scatter(x=df_transaction_train['t_dat'], y=df_transaction_train['price'], mode='lines+markers', name='Price')\nfig.add_trace(trace)\n\n# Get the minimum and maximum dates from the transaction table\nmin_date = df_transaction_train['t_dat'].min()\nmax_date = df_transaction_train['t_dat'].max()\n\n# Define step options for the slider\nstep_options = [\n    dict(method='relayout', args=['xaxis', {'range': [min_date, max_date]}], label='All'),\n    dict(method='relayout', args=['xaxis', {'range': [max_date - pd.DateOffset(years=1), max_date]}], label='Yearly'),\n    dict(method='relayout', args=['xaxis', {'range': [max_date - pd.DateOffset(months=1), max_date]}], label='Monthly'),\n    dict(method='relayout', args=['xaxis', {'range': [max_date - pd.DateOffset(weeks=1), max_date]}], label='Weekly'),\n    dict(method='relayout', args=['xaxis', {'range': [max_date - pd.DateOffset(days=1), max_date]}], label='Daily')\n]\n\n# Create a slider for adjusting the date range\ndate_slider = dict(\n    active=2,  # Index of the default step (Monthly)\n    currentvalue=dict(prefix=\"Time Interval: \"),\n    steps=[dict(label=step['label'], method=step['method'], args=step['args']) for step in step_options]\n)\n\n# Update layout to include the slider\nfig.update_layout(\n    sliders=[date_slider],\n    title=\"Sales Over Time with Date Range Slider\",\n    xaxis_title=\"Date\",\n    yaxis_title=\"Sales\"\n)\n\n# Show the interactive plot\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-13T09:30:24.217560Z","iopub.execute_input":"2023-10-13T09:30:24.218151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Story mode","metadata":{}},{"cell_type":"markdown","source":"*From the Transactions table:*\n1. *Sales Trends:* Analyze sales trends over time using the \"t_dat\" column. Identify peak sales periods and seasonal variations.\n\n2. *Customer Behavior:* Explore how often customers make purchases by analyzing the \"customer_id\" and \"t_dat\" columns. Identify the most active customers.\n\n3. *Price Analysis:* Investigate the distribution of prices and identify any price outliers or anomalies.\n\n4. *Sales Channels:* Compare sales performance between different sales channels (sales_channel_id) to understand which channels are more successful.\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#1 \nimport pandas as pd\nimport plotly.express as px\n\n\n\n# Convert the \"t_dat\" column to a datetime format\ndf_transaction_train['t_dat'] = pd.to_datetime(df_transaction_train['t_dat'])\n\n# Group transactions by date and calculate total sales for each day\ndaily_sales = df_transaction_train.groupby(df_transaction_train['t_dat'].dt.date)['price'].sum().reset_index()\n\n# Create an interactive line chart using Plotly\nfig = px.line(daily_sales, x='t_dat', y='price', title='Sales Trends Over Time')\nfig.update_xaxes(title_text='Date')\nfig.update_yaxes(title_text='Total Sales')\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T04:59:19.174157Z","iopub.execute_input":"2023-10-16T04:59:19.174527Z","iopub.status.idle":"2023-10-16T04:59:41.006159Z","shell.execute_reply.started":"2023-10-16T04:59:19.174491Z","shell.execute_reply":"2023-10-16T04:59:41.005097Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#2 \nimport pandas as pd\nimport plotly.express as px\n\n# Assuming you have a DataFrame named df_transaction_train\n\n\n\n# Group transactions by customer and count the number of purchases for each customer\ncustomer_purchase_counts = df_transaction_train.groupby('customer_id')['t_dat'].count().reset_index()\ncustomer_purchase_counts.columns = ['customer_id', 'purchase_count']\n\n# Sort customers by purchase count in descending order to identify the most active customers\nmost_active_customers = customer_purchase_counts.sort_values(by='purchase_count', ascending=False)\n\n# Display the top N most active customers\nN = 10  # You can change N to display more or fewer customers\ntop_active_customers = most_active_customers.head(N)\n\n# Create a bar chart to visualize customer behavior\nfig = px.bar(\n    top_active_customers,\n    x='customer_id',\n    y='purchase_count',\n    title='Top 10 Most Active Customers',\n    labels={'customer_id': 'Customer ID', 'purchase_count': 'Purchase Count'}\n)\n\nfig.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:01:57.923821Z","iopub.execute_input":"2023-10-16T05:01:57.924212Z","iopub.status.idle":"2023-10-16T05:02:09.587214Z","shell.execute_reply.started":"2023-10-16T05:01:57.924182Z","shell.execute_reply":"2023-10-16T05:02:09.586252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#3\nimport plotly.express as px\n\n# Assuming you have a DataFrame named df_transaction_train\n\n# Create a histogram to visualize the price distribution\nfig = px.histogram(\n    df_transaction_train,\n    x='price',\n    title='Price Distribution Analysis',\n    labels={'price': 'Price'},\n    nbins=50  # Adjust the number of bins as needed for your data\n)\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:03:31.275711Z","iopub.execute_input":"2023-10-16T05:03:31.276165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#4 \nimport plotly.express as px\n\n# Assuming you have a DataFrame named df_transaction_train\n\n# Create a bar chart to compare sales performance by sales channel\nfig = px.bar(\n    df_transaction_train,\n    x='sales_channel_id',\n    title='Sales Performance by Sales Channel',\n    labels={'sales_channel_id': 'Sales Channel', 'count': 'Number of Sales'},\n    color_discrete_sequence=px.colors.qualitative.Set1  # Adjust color scheme as desired\n)\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:15:53.063499Z","iopub.execute_input":"2023-10-16T05:15:53.063854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*From the Customers table:*\n5. *Customer Demographics:* Examine the distribution of customer ages to understand the age group of your customers.\n\n6. *Club Membership:* Analyze the distribution of club_member_status to see how many customers are part of the club.\n\n7. *Communication Frequency:* Investigate how often customers want to receive fashion news (fashion_news_frequency).\n","metadata":{}},{"cell_type":"code","source":"#6\nimport plotly.express as px\n\n# Calculate the count of customers in each club membership status\nclub_membership_counts = df_customers['club_member_status'].value_counts().reset_index()\nclub_membership_counts.columns = ['club_member_status', 'count']\n\n# Create a more attractive donut chart\nfig = px.pie(\n    club_membership_counts,\n    names='club_member_status',\n    values='count',\n    title='Distribution of Club Membership Status',\n    color_discrete_sequence=px.colors.qualitative.Pastel1,  # Choose a pleasant color scheme\n    hole=0.5,  # Increase the size of the central hole for a thicker donut chart\n    labels={'club_member_status': 'Club Membership Status'},\n)\n\n# Add some additional styling for aesthetics\nfig.update_traces(textposition='inside', textinfo='percent+label')\nfig.update_layout(\n    legend=dict(title='Status', x=0.9, y=0.5),\n    font=dict(family='Arial', size=14, color='black'),\n    margin=dict(t=30, l=30, r=30, b=30),\n    showlegend=True\n)\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:19:43.471319Z","iopub.execute_input":"2023-10-16T05:19:43.471982Z","iopub.status.idle":"2023-10-16T05:19:43.694995Z","shell.execute_reply.started":"2023-10-16T05:19:43.471938Z","shell.execute_reply":"2023-10-16T05:19:43.693862Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\n# Replace 'df_customers' and 'club_member_status' with your actual DataFrame and column names\nfig = px.pie(df_customers, names='club_member_status', title='Club Membership Distribution')\n\n# Combine categories with a low count into 'Other'\nthreshold = 100  # You can adjust this threshold as needed\ncategory_counts = df_customers['club_member_status'].value_counts()\ncategories_to_combine = category_counts[category_counts < threshold].index\ndf_customers['club_member_status_combined'] = df_customers['club_member_status'].apply(\n    lambda x: x if x not in categories_to_combine else 'Other'\n)\n\nfig = px.pie(df_customers, names='club_member_status_combined', title='Club Membership Distribution (with Other)')\n\nfig.update_traces(textinfo='percent+label')\n\n# Make it a donut chart by adjusting the 'hole' parameter\nfig.update_traces(hole=0.4)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:23:59.763873Z","iopub.execute_input":"2023-10-16T05:23:59.764293Z","iopub.status.idle":"2023-10-16T05:24:04.491483Z","shell.execute_reply.started":"2023-10-16T05:23:59.764259Z","shell.execute_reply":"2023-10-16T05:24:04.490032Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customers.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:36:36.379957Z","iopub.execute_input":"2023-10-16T05:36:36.380411Z","iopub.status.idle":"2023-10-16T05:36:36.396592Z","shell.execute_reply.started":"2023-10-16T05:36:36.380368Z","shell.execute_reply":"2023-10-16T05:36:36.395485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# found one anamoly which will be shifted to data cleaning ","metadata":{}},{"cell_type":"code","source":"df_customers['fashion_news_frequency'] = df_customers['fashion_news_frequency'].replace({'None':'NONE'})","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:40:25.578329Z","iopub.execute_input":"2023-10-16T05:40:25.578712Z","iopub.status.idle":"2023-10-16T05:40:25.807242Z","shell.execute_reply.started":"2023-10-16T05:40:25.578682Z","shell.execute_reply":"2023-10-16T05:40:25.806000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customers['fashion_news_frequency'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:40:37.686093Z","iopub.execute_input":"2023-10-16T05:40:37.686614Z","iopub.status.idle":"2023-10-16T05:40:37.771418Z","shell.execute_reply.started":"2023-10-16T05:40:37.686583Z","shell.execute_reply":"2023-10-16T05:40:37.770134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Sample DataFrame with customer data\ndata = {\n    'customer_id': df_customers['customer_id'].to_list(),\n    'fashion_news_frequency': df_customers['fashion_news_frequency'].to_list()\n}\n\ndf_customers = pd.DataFrame(data)\n\n# Step 1: Data Aggregation\nfrequency_counts = df_customers['fashion_news_frequency'].value_counts()\n\n# Step 2: Dynamic Grouping\n# Define dynamic groupings based on unique values\ngroupings = {\n    'Regularly': 'Frequent (Regularly)',\n    'Monthly': 'Infrequent (Monthly)',\n    np.nan: 'nan',\n    'NONE': 'No Fashion News'\n}\n\n# Group the data based on dynamic groupings\ndf_customers['dynamic_group'] = df_customers['fashion_news_frequency'].map(groupings)\n\n# Step 3: Visualization\nplt.figure(figsize=(10, 6))\nplt.bar(frequency_counts.index, frequency_counts.values, color='skyblue')\nplt.xlabel('Fashion News Frequency')\nplt.ylabel('Number of Customers')\nplt.title('Customer Fashion News Frequency Preferences')\nplt.xticks(rotation=45)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:41:20.236273Z","iopub.execute_input":"2023-10-16T05:41:20.236666Z","iopub.status.idle":"2023-10-16T05:41:21.039300Z","shell.execute_reply.started":"2023-10-16T05:41:20.236634Z","shell.execute_reply":"2023-10-16T05:41:21.038127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*From the Articles table:*\n\n8. *Product Analysis:* Analyze the distribution of product types (product_type_name) and their popularity.\n\n9. *Color Insights:* Explore the most common color groups and their perceived values.\n\n10. *Department and Category Analysis:* Investigate which departments and categories have the most articles and are most popular.\n\n11. *Text Analysis:* Perform text analysis on the \"detail_desc\" column to identify key features or attributes described in the articles.\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n\ndf_articles = df[['article_id','product_type_name']]\n\n# Step 1: Data Aggregation\nproduct_type_counts = df_articles['product_type_name'].value_counts()\n\n# Step 2: Visualization\nplt.figure(figsize=(23, 20))\nplt.bar(product_type_counts.index, product_type_counts.values, color='skyblue')\nplt.xlabel('Product Type')\nplt.ylabel('Number of Articles')\nplt.title('Distribution of Product Types')\nplt.xticks(rotation=90)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:45:01.718261Z","iopub.execute_input":"2023-10-16T05:45:01.718709Z","iopub.status.idle":"2023-10-16T05:45:03.030902Z","shell.execute_reply.started":"2023-10-16T05:45:01.718677Z","shell.execute_reply":"2023-10-16T05:45:03.029594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1: Data Aggregation\nproduct_type_counts = df_articles['product_type_name'].value_counts().reset_index()\nproduct_type_counts.columns = ['Product Type', 'Count']\n\n# Step 2: Visualization using Plotly\nfig = px.bar(product_type_counts, x='Product Type', y='Count', title='Distribution of Product Types')\nfig.update_traces(marker_color='skyblue')\nfig.update_layout(xaxis_title='Product Type', yaxis_title='Number of Articles')\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:45:57.812379Z","iopub.execute_input":"2023-10-16T05:45:57.812746Z","iopub.status.idle":"2023-10-16T05:45:57.884963Z","shell.execute_reply.started":"2023-10-16T05:45:57.812716Z","shell.execute_reply":"2023-10-16T05:45:57.883927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#3\nimport pandas as pd\nimport plotly.express as px\n\n\n\ndf_articles = df[['article_id','colour_group_name','perceived_colour_value_name']]\n\n# Step 1: Data Aggregation\ncolor_group_counts = df_articles['colour_group_name'].value_counts().reset_index()\ncolor_group_counts.columns = ['Color Group', 'Count']\n\nperceived_value_counts = df_articles['perceived_colour_value_name'].value_counts().reset_index()\nperceived_value_counts.columns = ['Perceived Value', 'Count']\n\n# Step 2: Visualization using Plotly\nfig = px.bar(color_group_counts, x='Color Group', y='Count', title='Distribution of Color Groups')\nfig.update_traces(marker_color='skyblue')\nfig.update_layout(xaxis_title='Color Group', yaxis_title='Number of Articles')\n\nfig.show()\n\nfig2 = px.bar(perceived_value_counts, x='Perceived Value', y='Count', title='Distribution of Perceived Values')\nfig2.update_traces(marker_color='lightcoral')\nfig2.update_layout(xaxis_title='Perceived Value', yaxis_title='Number of Articles')\n\nfig2.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:47:28.563641Z","iopub.execute_input":"2023-10-16T05:47:28.564062Z","iopub.status.idle":"2023-10-16T05:47:28.718496Z","shell.execute_reply.started":"2023-10-16T05:47:28.564028Z","shell.execute_reply":"2023-10-16T05:47:28.717268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#10 \nimport pandas as pd\nimport plotly.express as px\n\n\ndf_articles = df[['article_id','department_name','index_name']]\n\n# Step 1: Data Aggregation\ndepartment_counts = df_articles['department_name'].value_counts().reset_index()\ndepartment_counts.columns = ['Department', 'Count']\n\ncategory_counts = df_articles['index_name'].value_counts().reset_index()\ncategory_counts.columns = ['Category', 'Count']\n\n# Step 2: Visualization using Plotly\nfig = px.bar(department_counts, x='Department', y='Count', title='Number of Articles in Each Department')\nfig.update_traces(marker_color='skyblue')\nfig.update_layout(xaxis_title='Department', yaxis_title='Number of Articles')\n\nfig.show()\n\nfig2 = px.bar(category_counts, x='Category', y='Count', title='Number of Articles in Each Category')\nfig2.update_traces(marker_color='lightcoral')\nfig2.update_layout(xaxis_title='Category', yaxis_title='Number of Articles')\n\nfig2.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:48:59.120888Z","iopub.execute_input":"2023-10-16T05:48:59.121378Z","iopub.status.idle":"2023-10-16T05:48:59.261666Z","shell.execute_reply.started":"2023-10-16T05:48:59.121339Z","shell.execute_reply":"2023-10-16T05:48:59.259477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport nltk\nfrom nltk.corpus import stopwords\nfrom wordcloud import WordCloud\nimport matplotlib.pyplot as plt\n\n\n\ndf_articles = df[['article_id','detail_desc']]\n\n# 1. Text Preprocessing\nnltk.download('stopwords')\nstop_words = set(stopwords.words('english'))\n\ndef preprocess_text(text):\n    text = text.lower()\n    text = \" \".join([word for word in text.split() if word not in stop_words])\n    text = text.replace('.', '')  # Remove periods\n    return text\n\ndf_articles['cleaned_desc'] = df_articles['detail_desc'].apply(preprocess_text)\n\n# 2. Tokenization\ntokenized_desc = df_articles['cleaned_desc'].str.split()\n\n# 3. Frequency Analysis\nword_freq = nltk.FreqDist([word for desc in tokenized_desc for word in desc])\n\n# Print the most common words\nprint(word_freq.most_common(10))\n\n# 4. Word Cloud\nwordcloud = WordCloud(width=800, height=400, background_color='white').generate_from_frequencies(word_freq)\n\nplt.figure(figsize=(10, 5))\nplt.imshow(wordcloud, interpolation='bilinear')\nplt.axis('off')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:50:26.282350Z","iopub.execute_input":"2023-10-16T05:50:26.282773Z","iopub.status.idle":"2023-10-16T05:50:28.319521Z","shell.execute_reply.started":"2023-10-16T05:50:26.282739Z","shell.execute_reply":"2023-10-16T05:50:28.317879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# import pandas as pd\n# import nltk\n# from nltk.corpus import stopwords\n# from wordcloud import WordCloud\n# import matplotlib.pyplot as plt\n\n\n\n# df_articles = df[['article_id','detail_desc']]\n# # 1. Text Preprocessing\n# nltk.download('stopwords')\n# stop_words = set(stopwords.words('english'))\n\n# def preprocess_text(text):\n#     if text is not None:  # Check for NaN (missing) values\n#         text = text.lower()\n#         text = \" \".join([word for word in text.split() if word not in stop_words])\n#         text = text.replace('.', '')  # Remove periods\n#     return text\n\n# df_articles['cleaned_desc'] = df_articles['detail_desc'].apply(preprocess_text)\n\n# # 2. Tokenization\n# tokenized_desc = df_articles['cleaned_desc'].str.split()\n\n# # 3. Frequency Analysis\n# word_freq = nltk.FreqDist([word for desc in tokenized_desc for word in desc])\n\n# # Print the most common words\n# print(word_freq.most_common(10))\n\n# # 4. Word Cloud\n# wordcloud = WordCloud(width=800, height=400, background_color='white').generate_from_frequencies(word_freq)\n\n# plt.figure(figsize=(10, 5))\n# plt.imshow(wordcloud, interpolation='bilinear')\n# plt.axis('off')\n# plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T05:53:38.920184Z","iopub.execute_input":"2023-10-16T05:53:38.920596Z","iopub.status.idle":"2023-10-16T05:53:39.032251Z","shell.execute_reply.started":"2023-10-16T05:53:38.920564Z","shell.execute_reply":"2023-10-16T05:53:39.030706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport nltk\nfrom nltk.corpus import stopwords\nfrom wordcloud import WordCloud\nimport matplotlib.pyplot as plt\n\ndf_articles = df[['article_id','detail_desc']]\n\n\n\n# 2. Text Preprocessing\nnltk.download('stopwords')\nstop_words = set(stopwords.words('english'))\n\ndef preprocess_text(text):\n    if isinstance(text, str):\n        text = text.lower()\n        text = \" \".join([word for word in text.split() if word not in stop_words])\n        text = text.replace('.', '')  # Remove periods\n    return text\n\n# Apply text preprocessing\ndf_articles['cleaned_desc'] = df_articles['detail_desc'].apply(preprocess_text)\n\n# 3. Tokenization\ntokenized_desc = df_articles['cleaned_desc'].str.split()\n\n\n# 4. Frequency Analysis\nword_freq = nltk.FreqDist(word for desc in tokenized_desc for word in desc if isinstance(word, str) and word.isalpha())\n\n\n\n# # Check if there are words to create a word cloud\n# if len(word_freq) > 0:\n#     # 5. Word Cloud\n#     wordcloud = WordCloud(width=800, height=400, background_color='white').generate_from_frequencies(word_freq)\n\n#     plt.figure(figsize=(10, 5))\n#     plt.imshow(wordcloud, interpolation='bilinear')\n#     plt.axis('off')\n#     plt.show()\n# else:\n#     print(\"No valid words to create a word cloud.\")\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T06:14:20.143824Z","iopub.execute_input":"2023-10-16T06:14:20.144217Z","iopub.status.idle":"2023-10-16T06:14:21.917826Z","shell.execute_reply.started":"2023-10-16T06:14:20.144186Z","shell.execute_reply":"2023-10-16T06:14:21.916255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nfrom wordcloud import WordCloud\nimport matplotlib.pyplot as plt\n\n# Ensure NLTK stopwords are downloaded\nnltk.download('stopwords')\n\n# List of English stopwords\nstop_words = nltk.corpus.stopwords.words('english')\n\n# Function to preprocess text\ndef preprocess_text(text):\n    if isinstance(text, str):\n        text = text.lower()\n        # Remove non-alphabet characters and split into words\n        words = [word for word in text.split() if word.isalpha()]\n        return \" \".join(words)\n    return \"\"\n\n# Apply text preprocessing to the 'detail_desc' column\ndf_articles['cleaned_desc'] = df_articles['detail_desc'].apply(preprocess_text)\n\n# Tokenization\ntokenized_desc = df_articles['cleaned_desc'].str.split()\n\n# Frequency Analysis\nword_freq = nltk.FreqDist(word for desc in tokenized_desc for word in desc)\n\n# Filter out low-frequency words\nmin_word_freq = 5\nfiltered_word_freq = {word: freq for word, freq in word_freq.items() if freq >= min_word_freq}\n\n# Print the most common words\nprint(list(filtered_word_freq.items())[:10])\n\n# Word Cloud\nwordcloud = WordCloud(width=800, height=400, background_color='white').generate_from_frequencies(filtered_word_freq)\n\nplt.figure(figsize=(10, 5))\nplt.imshow(wordcloud, interpolation='bilinear')\nplt.axis(\"off\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T06:16:38.806945Z","iopub.execute_input":"2023-10-16T06:16:38.807346Z","iopub.status.idle":"2023-10-16T06:16:41.997820Z","shell.execute_reply.started":"2023-10-16T06:16:38.807313Z","shell.execute_reply":"2023-10-16T06:16:41.997058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}