{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"},{"sourceId":10304978,"sourceType":"datasetVersion","datasetId":6378869},{"sourceId":10314637,"sourceType":"datasetVersion","datasetId":6385606},{"sourceId":10325800,"sourceType":"datasetVersion","datasetId":6393394},{"sourceId":10325857,"sourceType":"datasetVersion","datasetId":6393428}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/0 (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt # graph visualization \nimport matplotlib.image as mpimg # image visualizations \nfrom IPython.display import Image, display # image visualizations \nimport seaborn as sns # graph visualizations\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-15T14:54:45.393173Z","iopub.execute_input":"2025-01-15T14:54:45.393474Z","iopub.status.idle":"2025-01-15T14:59:38.24065Z","shell.execute_reply.started":"2025-01-15T14:54:45.393445Z","shell.execute_reply":"2025-01-15T14:59:38.23978Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')                   \ncustomers = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\nsample_submission =pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\ntransactions_train =pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T14:59:38.241986Z","iopub.execute_input":"2025-01-15T14:59:38.242305Z","iopub.status.idle":"2025-01-15T15:01:11.665626Z","shell.execute_reply.started":"2025-01-15T14:59:38.242284Z","shell.execute_reply":"2025-01-15T15:01:11.664589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\ndf_s = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv') \ndf_s","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:11.702225Z","iopub.execute_input":"2025-01-15T15:01:11.702583Z","iopub.status.idle":"2025-01-15T15:01:14.477Z","shell.execute_reply.started":"2025-01-15T15:01:11.702554Z","shell.execute_reply":"2025-01-15T15:01:14.475962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_s.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:14.477964Z","iopub.execute_input":"2025-01-15T15:01:14.478296Z","iopub.status.idle":"2025-01-15T15:01:14.628983Z","shell.execute_reply.started":"2025-01-15T15:01:14.478254Z","shell.execute_reply":"2025-01-15T15:01:14.627911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_s= df_s.head(1000)\ndf_s","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:14.630082Z","iopub.execute_input":"2025-01-15T15:01:14.63043Z","iopub.status.idle":"2025-01-15T15:01:14.640755Z","shell.execute_reply.started":"2025-01-15T15:01:14.630393Z","shell.execute_reply":"2025-01-15T15:01:14.639875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_s.prediction[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:14.643334Z","iopub.execute_input":"2025-01-15T15:01:14.643589Z","iopub.status.idle":"2025-01-15T15:01:14.660595Z","shell.execute_reply.started":"2025-01-15T15:01:14.643562Z","shell.execute_reply":"2025-01-15T15:01:14.659812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Histogramme des prédictions\ndf_s['prediction'].hist()\nplt.title('Distribution des Prédictions')\nplt.xlabel('Prédiction')\nplt.ylabel('Fréquence')\nplt.show()\n\n# Boxplot des prédictions par client\nsns.boxplot(x='customer_id', y='prediction', data=df)\nplt.title('Boxplot des Prédictions par Client')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:14.661662Z","iopub.execute_input":"2025-01-15T15:01:14.661914Z","iopub.status.idle":"2025-01-15T15:01:15.101663Z","shell.execute_reply.started":"2025-01-15T15:01:14.661882Z","shell.execute_reply":"2025-01-15T15:01:15.096631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\ntransactions =pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.102005Z","iopub.status.idle":"2025-01-15T15:01:15.102237Z","shell.execute_reply":"2025-01-15T15:01:15.102142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.102899Z","iopub.status.idle":"2025-01-15T15:01:15.103376Z","shell.execute_reply":"2025-01-15T15:01:15.103202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.104391Z","iopub.status.idle":"2025-01-15T15:01:15.104772Z","shell.execute_reply":"2025-01-15T15:01:15.104604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Since count is not shown it means there are no null values in the table\ntransactions.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.10556Z","iopub.status.idle":"2025-01-15T15:01:15.105969Z","shell.execute_reply":"2025-01-15T15:01:15.105788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the minimum and maximum values in a column\ncolumn_min = transactions['price'].min()\ncolumn_max = transactions['price'].max()\n\n# Convert numbers to strings and format without scientific notation\nformatted_min = '{:.6f}'.format(column_min)\nformatted_max = '{:.6f}'.format(column_max)\n\nprint(f\"Minimum value: {formatted_min}\")\nprint(f\"Maximum value: {formatted_max}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.106914Z","iopub.status.idle":"2025-01-15T15:01:15.107267Z","shell.execute_reply":"2025-01-15T15:01:15.107111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.float_format', '{:.5f}'.format)\ntransactions.describe()['price']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.108138Z","iopub.status.idle":"2025-01-15T15:01:15.10852Z","shell.execute_reply":"2025-01-15T15:01:15.108344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Observe outliers\nplt.figure(figsize=(10, 5))\nsns.violinplot(data=transactions['price'])\nplt.xlabel('Column')\nplt.ylabel('Values')\nplt.title('Violin Plot of price')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.109415Z","iopub.status.idle":"2025-01-15T15:01:15.109839Z","shell.execute_reply":"2025-01-15T15:01:15.109667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the start, end, and width of the range\nstart = 0.0\nend = 0.6\nwidth = 0.05\n\n# Generate the range labels\nlabels = [f\"< {start+width:.3f}\"]\nlabels.extend([f\"{i:.3f} - {(i+width):.3f}\" for i in np.arange(start+width, end, width)])\nlabels.append(f\"> {end:.3f}\")\n\n# Create the bins with labels\nbins = [start] + [i+width for i in np.arange(start, end, width)] + [float('inf')]\n\n# Create a new column with the price ranges\ntransactions['price_range'] = pd.cut(transactions['price'], bins=bins, labels=labels, right=False)\n\n# Count the number of prices in each range\nprice_counts = transactions['price_range'].value_counts().sort_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.110566Z","iopub.status.idle":"2025-01-15T15:01:15.11101Z","shell.execute_reply":"2025-01-15T15:01:15.110824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate percentages\ntotal_count = price_counts.sum()\npercentages = (price_counts / total_count) * 100\n\n# Format the count and percentage columns\nformatted_counts = price_counts.map(\"{:,}\".format)\nformatted_percentages = percentages.map(\"{:.2f}%\".format)\n\nresult_df = pd.DataFrame({'Count': formatted_counts, 'Percentage': formatted_percentages})\nprint(result_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.111809Z","iopub.status.idle":"2025-01-15T15:01:15.112172Z","shell.execute_reply":"2025-01-15T15:01:15.112015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.112758Z","iopub.status.idle":"2025-01-15T15:01:15.113111Z","shell.execute_reply":"2025-01-15T15:01:15.112957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#articles \narticles.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.113836Z","iopub.status.idle":"2025-01-15T15:01:15.1142Z","shell.execute_reply":"2025-01-15T15:01:15.114036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.114749Z","iopub.status.idle":"2025-01-15T15:01:15.115097Z","shell.execute_reply":"2025-01-15T15:01:15.114934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.nunique() #number of unique values for eah colum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.115765Z","iopub.status.idle":"2025-01-15T15:01:15.116136Z","shell.execute_reply":"2025-01-15T15:01:15.116009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.isnull().sum() #the number of misssing   values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.116898Z","iopub.status.idle":"2025-01-15T15:01:15.117262Z","shell.execute_reply":"2025-01-15T15:01:15.117097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles['product_type_name'].value_counts()[:20].plot(kind='barh') #on a filtrer sur 20 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.118019Z","iopub.status.idle":"2025-01-15T15:01:15.118387Z","shell.execute_reply":"2025-01-15T15:01:15.118223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Percentage of null values in a column: \", (100 * articles['detail_desc'].isna().sum() / articles.shape[0]).round(4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.119126Z","iopub.status.idle":"2025-01-15T15:01:15.119507Z","shell.execute_reply":"2025-01-15T15:01:15.119332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.apply(lambda x: x.nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.120154Z","iopub.status.idle":"2025-01-15T15:01:15.120541Z","shell.execute_reply":"2025-01-15T15:01:15.120363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# groups and returns count of appearance\ndef find_values_with_multiple_references(df, column2, column3):\n    grouped = df.groupby(column2)[column3].nunique()\n    result = df[df[column2]\n                .isin(grouped[grouped > 1].index)] \\\n        .groupby([column2, column3]) \\\n        .size() \\\n        .reset_index(name='count')\n    \n    return result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.121216Z","iopub.status.idle":"2025-01-15T15:01:15.121608Z","shell.execute_reply":"2025-01-15T15:01:15.121421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now let's check some columns\nprint(find_values_with_multiple_references(articles, 'product_type_name', 'product_type_no'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.122445Z","iopub.status.idle":"2025-01-15T15:01:15.122995Z","shell.execute_reply":"2025-01-15T15:01:15.122802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[articles['product_type_name'] == 'Umbrella'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.123714Z","iopub.status.idle":"2025-01-15T15:01:15.124085Z","shell.execute_reply":"2025-01-15T15:01:15.123927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[(articles['department_name'] == 'Accessories') & (articles['department_no'] == 3510)].head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.12492Z","iopub.status.idle":"2025-01-15T15:01:15.125188Z","shell.execute_reply":"2025-01-15T15:01:15.125079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[(articles['department_name'] == 'Accessories') & (articles['department_no'] == 3941)].head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.126653Z","iopub.status.idle":"2025-01-15T15:01:15.126913Z","shell.execute_reply":"2025-01-15T15:01:15.126811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(find_values_with_multiple_references(articles, 'section_name', 'section_no'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.128266Z","iopub.status.idle":"2025-01-15T15:01:15.128634Z","shell.execute_reply":"2025-01-15T15:01:15.128442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[articles['section_name'] == 'Ladies Other']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.129588Z","iopub.status.idle":"2025-01-15T15:01:15.129835Z","shell.execute_reply":"2025-01-15T15:01:15.129734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert multiple columns to category type\ncolumns_to_convert = ['graphical_appearance_name', 'colour_group_name', 'perceived_colour_value_name', 'perceived_colour_master_name',\n                     'index_name', 'index_group_name', 'garment_group_name']\n# Assuming 'df' is your DataFrame and 'columns' is a list of column names\narticles[columns_to_convert] = convert_to_category(articles, columns=columns_to_convert)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.130389Z","iopub.status.idle":"2025-01-15T15:01:15.13068Z","shell.execute_reply":"2025-01-15T15:01:15.130559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Since index code is string we will enumerate it so that index_name column has its enumerated ids\narticles['index_id'] = (articles['index_code'].astype('category').cat.codes + 100).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.131465Z","iopub.status.idle":"2025-01-15T15:01:15.131794Z","shell.execute_reply":"2025-01-15T15:01:15.131689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the number of numerical, text, and categorical columns\ndef count_numerical_text_categorical_columns(df):\n    numerical_cols = []\n    text_cols = []\n    categorical_cols = []\n\n    for column in df.columns:\n        if pd.api.types.is_numeric_dtype(df[column]):\n            numerical_cols.append(column)\n        elif pd.api.types.is_string_dtype(df[column]):\n            text_cols.append(column)\n        elif pd.api.types.is_categorical_dtype(df[column]):\n            categorical_cols.append(column)\n\n    return (\n        len(numerical_cols),\n        len(text_cols),\n        len(categorical_cols),\n        numerical_cols,\n        text_cols,\n        categorical_cols\n    )\n\n# Example usage\n(\n    num_numerical_cols,\n    num_text_cols,\n    num_categorical_cols,\n    numerical_cols,\n    text_cols,\n    categorical_cols\n) = count_numerical_text_categorical_columns(articles)\n\nprint(f\"Number of numerical columns: {num_numerical_cols}\")\nprint(f\"Numerical columns: {numerical_cols}\")\nprint('\\n' + 100 * \"=\" + '\\n')\nprint(f\"Number of text columns: {num_text_cols}\")\nprint(f\"Text columns: {text_cols}\")\nprint('\\n' + 100 * \"=\" + '\\n')\nprint(f\"Number of categorical columns: {num_categorical_cols}\")\nprint(f\"Categorical columns: {categorical_cols}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.132575Z","iopub.status.idle":"2025-01-15T15:01:15.132877Z","shell.execute_reply":"2025-01-15T15:01:15.132739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count the occurrences of each category\ncategory_counts = articles['index_name'].value_counts()\n\n# Sort the categories by count in descending order\nsorted_categories = category_counts.sort_values(ascending=False).index\n\n# Set the color palette\ncolors = sns.color_palette('Set3', len(sorted_categories))\n\n# Plot the histogram\nsns.set(style='ticks')\nplt.figure(figsize=(8, 6))\nsns.countplot(x='index_name', data=articles, order=sorted_categories, palette=colors)\nplt.xlabel('Category')\nplt.ylabel('Count')\nplt.title('Histogram of Categories (Sorted by Count)')\n\n# Rotate x-axis labels\nplt.xticks(rotation=90, size=9)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.133765Z","iopub.status.idle":"2025-01-15T15:01:15.134051Z","shell.execute_reply":"2025-01-15T15:01:15.133947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the counts for each combination\ncounts = articles.groupby(['garment_group_name', 'index_group_name']).size().unstack(fill_value=0)\n\n# Calculate the total count for each garment group\ntotal_counts = counts.sum(axis=1)\n\n# Sort the values in descending order by the total count\nsorted_counts = counts.loc[total_counts.sort_values(ascending=True).index]\n\n# Create the plot\nplt.figure(figsize=(18, 12))\nsorted_counts.plot(kind='barh', stacked=True, color=sns.color_palette('Set3'))\nplt.xlabel('Index Group')\nplt.ylabel('Count')\nplt.title('Relationship between index_group_name and garment_group_name')\n\n# Display the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.134715Z","iopub.status.idle":"2025-01-15T15:01:15.135027Z","shell.execute_reply":"2025-01-15T15:01:15.134897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"counts = articles.groupby(['index_group_name', 'index_name']).count()['article_id']\nnon_zero_counts = counts[counts > 0]\nprint(non_zero_counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.135786Z","iopub.status.idle":"2025-01-15T15:01:15.136132Z","shell.execute_reply":"2025-01-15T15:01:15.136005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.options.display.max_rows = None\n\n# Group by 'product_group_name' and 'product_type_name' and count 'article_id'\ncounts = articles.groupby(['product_group_name', 'product_type_name'])['article_id'].count()\n\n# Sort within each main group\nsorted_counts = counts.groupby(level=0, group_keys=False).apply(lambda x: x.sort_values(ascending=False))\n\n# Print the resulting counts\nprint(sorted_counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.136909Z","iopub.status.idle":"2025-01-15T15:01:15.137153Z","shell.execute_reply":"2025-01-15T15:01:15.137056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#customers","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.138129Z","iopub.status.idle":"2025-01-15T15:01:15.138479Z","shell.execute_reply":"2025-01-15T15:01:15.138345Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"customers.head() #affichage des premières lignes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.139479Z","iopub.status.idle":"2025-01-15T15:01:15.139865Z","shell.execute_reply":"2025-01-15T15:01:15.139728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"customers.shape #affichage de les dimensions de la data ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.14067Z","iopub.status.idle":"2025-01-15T15:01:15.140951Z","shell.execute_reply":"2025-01-15T15:01:15.140824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if customer_ids are unique :sert à comparer le nombre total d'enregistrements (lignes) dans le DataFrame customers avec le nombre d'identifiants clients uniques. \ncustomers.shape[0] - customers['customer_id'].nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.141826Z","iopub.status.idle":"2025-01-15T15:01:15.142118Z","shell.execute_reply":"2025-01-15T15:01:15.142015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Customer club status: \", customers['club_member_status'].unique())#Afficher les valeurs uniques dans la colonne club_member_status du DataFrame customers.\nprint(\"Customer fashion news receiver frequency: \", customers['fashion_news_frequency'].unique())#Afficher les valeurs uniques dans la colonne fashion_news_frequency, qui pourrait contenir la fréquence à laquelle un client reçoit des nouvelles de mode.\nprint(\"Customer age: \", customers['age'].unique()) # Afficher les valeurs uniques dans la colonne age, qui représente l'âge des clients.\nprint(\"Customer location: \", customers['postal_code'].nunique()) #Afficher le nombre de codes postaux uniques dans la colonne postal_code.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.142913Z","iopub.status.idle":"2025-01-15T15:01:15.143242Z","shell.execute_reply":"2025-01-15T15:01:15.143085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# L'objectif est de remplacer toutes ces valeurs par une valeur unique 'None', afin de standardiser et simplifier la colonne pour l'analyse ou le traitement.\ncustomers.loc[~customers['fashion_news_frequency'].isin(['Regularly', 'Monthly']), 'fashion_news_frequency'] = 'None'\ncustomers['fashion_news_frequency'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.144277Z","iopub.status.idle":"2025-01-15T15:01:15.144629Z","shell.execute_reply":"2025-01-15T15:01:15.144445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dentifier les codes postaux avec le plus grand nombre de clients dans le DataFrame customers.\ncust_location = customers[['postal_code', 'customer_id']] \\\n    .groupby('postal_code', as_index=False) \\\n    .count() \\\n    .sort_values('customer_id', ascending=False)\ncust_location.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.145353Z","iopub.status.idle":"2025-01-15T15:01:15.145694Z","shell.execute_reply":"2025-01-15T15:01:15.14557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#visualiser la distribution des âges des clients à l'aide d'un histogramme, en utilisant la bibliothèque Seaborn pour le style graphique et Matplotlib pour personnaliser davantage l'affichage.\nsns.set_style(\"ticks\")#Configure le style graphique de Seaborn pour utiliser le style \"ticks\", ce qui ajoute des petites graduations aux axes et rend le graphique esthétiquement plus clair\nf, ax = plt.subplots(figsize=(10,5))#Crée une figure et un axe avec des dimensions spécifiques (10 pouces de largeur et 5 pouces de hauteur).\nax = sns.histplot(data=customers, x='age', bins=50, color='#82cbb2')# Trace un histogramme basé sur les données de la colonne age du DataFrame customers\nax.set_xlabel('Distribution of the customers age')\n\n# Set x-axis label format for every 10 years\nax.xaxis.set_major_locator(ticker.MultipleLocator(base=10))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.14646Z","iopub.status.idle":"2025-01-15T15:01:15.146818Z","shell.execute_reply":"2025-01-15T15:01:15.146696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#visualiser la distribution des statuts des membres du club (variable club_member_status) à l'aide d'un histogramme\nf, ax = plt.subplots(figsize=(10,5))\nax = sns.histplot(data=customers, x='club_member_status', color='#8e82fe')\nax.set_xlabel('Distribution of club member status')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.147696Z","iopub.status.idle":"2025-01-15T15:01:15.147976Z","shell.execute_reply":"2025-01-15T15:01:15.147861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#L'objectif de ce code est de visualiser la distribution des statuts des membres du club (club_member_status) en utilisant un graphique circulaire (camembert).\nsns.set_style(\"darkgrid\")#Définit le style graphique global sur \"darkgrid\", qui applique un fond sombre avec des lignes de grille légères pour une meilleure lisibilité.\nf, ax = plt.subplots(figsize=(10, 5))#Crée une figure et un axe avec une taille personnalisée (10 pouces de largeur et 5 pouces de hauteur).\ncolors = sns.color_palette('Set3')\nid_news = customers[['customer_id', 'club_member_status']].groupby('club_member_status')['customer_id'].count()\n\n# Calculate percentages\npercentages = id_news / id_news.sum() * 100\n\n# Create the pie chart with percentage labels\nax.pie(id_news, labels=id_news.index, colors=colors, autopct='%1.2f%%')\nax.set_facecolor('lightgrey')\nax.set_xlabel('Distribution of fashion news frequency')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.149109Z","iopub.status.idle":"2025-01-15T15:01:15.149425Z","shell.execute_reply":"2025-01-15T15:01:15.149315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# When it comes to fashion news interest people are either not interested at all or they want a regular update\n# with two-thirds choosing not to receive any news\n\nsns.set_style(\"darkgrid\")\nf, ax = plt.subplots(figsize=(10, 5))\ncolors = sns.color_palette('Set3')\nid_news = customers[['customer_id', 'fashion_news_frequency']].groupby('fashion_news_frequency')['customer_id'].count()\n\n# Calculate percentages\npercentages = id_news / id_news.sum() * 100\n\n# Create the pie chart with percentage labels\nax.pie(id_news, labels=id_news.index, colors=colors, autopct='%1.2f%%')\nax.set_facecolor('lightgrey')\nax.set_xlabel('Distribution of fashion news frequency')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.150019Z","iopub.status.idle":"2025-01-15T15:01:15.150257Z","shell.execute_reply":"2025-01-15T15:01:15.150159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Group customer IDs and count purchases:visualiser la distribution du nombre d'achats effectués par les clients à l'aide d'un histogramme\npurchase_counts = transactions['customer_id'].value_counts()#purchase_counts, est une série pandas où les indices représentent les customer_id et les valeurs indiquent le nombre d'achats associés.\n\n# Plot the histogram of purchase counts\nplt.figure(figsize=(10, 6))\nplt.hist(purchase_counts, bins=range(1, purchase_counts.max()+2), edgecolor='black')\nplt.xlabel('Number of Purchases')\nplt.ylabel('Frequency')\nplt.title('Distribution of Purchases')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.151106Z","iopub.status.idle":"2025-01-15T15:01:15.151452Z","shell.execute_reply":"2025-01-15T15:01:15.151284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#visualiser la distribution des achats, mais avec une échelle logarithmique pour l'axe des x.\nplt.figure(figsize=(10, 6))\nplt.hist(purchase_counts, bins=range(1, purchase_counts.max()+2), color='#464196', edgecolor='#8f8ce7')\nplt.xscale('log')  # Set x-axis scale to logarithmic\nplt.xlabel('Number of Purchases')\nplt.ylabel('Frequency')\nplt.title('Distribution of Purchases (Log Scale)')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.152409Z","iopub.status.idle":"2025-01-15T15:01:15.15273Z","shell.execute_reply":"2025-01-15T15:01:15.152602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Transactions des achats ","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-01-15T15:01:15.153607Z","iopub.status.idle":"2025-01-15T15:01:15.153918Z","shell.execute_reply":"2025-01-15T15:01:15.153805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#transactions des achats\nperiod_len = transactions['t_dat'].nunique()#donne la durée de la période pendant laquelle les achats ont été effectués\nstart_date = transactions['t_dat'].min()#c'est-à-dire la première date à laquelle un achat a eu lieu.\nend_date = transactions['t_dat'].max()#c'est-à-dire la dernière date à laquelle un achat a eu lieu.\nprint(\"Purchase history period length: \", period_len)\nprint(\"Purchase history start date: \", start_date)\nprint(\"Purchase history end date: \", end_date)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.154515Z","iopub.status.idle":"2025-01-15T15:01:15.154768Z","shell.execute_reply":"2025-01-15T15:01:15.154668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert string dates to date objects\ndate1 = datetime.strptime(start_date, '%Y-%m-%d').date()\ndate2 = datetime.strptime(end_date, '%Y-%m-%d').date()\n\n# Calculate the number of days between the dates\ndays_between = (date2 - date1).days + 1\n\nprint(f\"Number of days between {start_date} and {end_date}: {days_between}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.155599Z","iopub.status.idle":"2025-01-15T15:01:15.15587Z","shell.execute_reply":"2025-01-15T15:01:15.155759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\ninput_dir = Path(\"../input/h-and-m-personalized-fashion-recommendations\")\nlist(input_dir.iterdir())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.156604Z","iopub.status.idle":"2025-01-15T15:01:15.156885Z","shell.execute_reply":"2025-01-15T15:01:15.156764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install polars --upgrade -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.157592Z","iopub.status.idle":"2025-01-15T15:01:15.157899Z","shell.execute_reply":"2025-01-15T15:01:15.157756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\npl.show_versions()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.158889Z","iopub.status.idle":"2025-01-15T15:01:15.159165Z","shell.execute_reply":"2025-01-15T15:01:15.159059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"customers = pl.read_csv(Path(input_dir) / \"customers.csv\")\nprint(customers.shape)\ncustomers.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.159804Z","iopub.status.idle":"2025-01-15T15:01:15.160044Z","shell.execute_reply":"2025-01-15T15:01:15.159947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions = pl.read_csv(Path(input_dir) / \"transactions_train.csv\")\nprint(transactions.shape)\ntransactions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.160638Z","iopub.status.idle":"2025-01-15T15:01:15.160871Z","shell.execute_reply":"2025-01-15T15:01:15.160776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles = pl.read_csv(Path(input_dir) / \"articles.csv\")\nprint(articles.shape)\narticles.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.161445Z","iopub.status.idle":"2025-01-15T15:01:15.161738Z","shell.execute_reply":"2025-01-15T15:01:15.161633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission = pl.read_csv(Path(input_dir) / \"sample_submission.csv\")\nprint(sample_submission.shape)\nsample_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.162262Z","iopub.status.idle":"2025-01-15T15:01:15.162521Z","shell.execute_reply":"2025-01-15T15:01:15.162398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions = transactions.with_columns(\n    pl.col(\"t_dat\").str.to_datetime()\n).with_columns(\n    pl.col(\"t_dat\").dt.weekday().alias(\"weekday\")\n)\ntransactions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.163045Z","iopub.status.idle":"2025-01-15T15:01:15.163288Z","shell.execute_reply":"2025-01-15T15:01:15.163187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"t_min, t_max = transactions[\"t_dat\"].min(), transactions[\"t_dat\"].max()\nduration = t_max - t_min\nduration","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.163846Z","iopub.status.idle":"2025-01-15T15:01:15.164091Z","shell.execute_reply":"2025-01-15T15:01:15.163991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = transactions.filter(\n    pl.col(\"t_dat\") <= t_max - duration/3\n)\ntest = transactions.filter(\n    pl.col(\"t_dat\") > t_max - duration/3\n)\ntrain.shape, test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.164775Z","iopub.status.idle":"2025-01-15T15:01:15.165014Z","shell.execute_reply":"2025-01-15T15:01:15.164915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Only make predictions on last month of data\nlast_month = t_max - duration/3 - pl.duration(weeks=4)\ntrain_last_month = train.filter(\n    pl.col(\"t_dat\") > last_month\n)\ntrain_last_month.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.165667Z","iopub.status.idle":"2025-01-15T15:01:15.165938Z","shell.execute_reply":"2025-01-15T15:01:15.165827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Quick preprocessing\ndef dummies_and_cast(df):\n    return pl.concat([\n        df[[\"t_dat\", \"customer_id\", \"article_id\", \"price\"]],\n        df[\"weekday\"].to_dummies(),\n        df[\"sales_channel_id\"].cast(pl.Utf8).to_frame(),\n    ], how=\"horizontal\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.166676Z","iopub.status.idle":"2025-01-15T15:01:15.166946Z","shell.execute_reply":"2025-01-15T15:01:15.16682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_last_month = dummies_and_cast(train_last_month)\ntrain_last_month.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.167727Z","iopub.status.idle":"2025-01-15T15:01:15.168048Z","shell.execute_reply":"2025-01-15T15:01:15.167929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#K-means and feature importance for articles","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.168709Z","iopub.status.idle":"2025-01-15T15:01:15.168965Z","shell.execute_reply":"2025-01-15T15:01:15.168846Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#K-means and feature importance for articles\n#Load and group data\ntransactions = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', parse_dates=['t_dat'])\ntransactions['customer_id'] = transactions['customer_id'].str[-16:].str.hex_to_int().astype('int64')\ntransactions['article_id'] = transactions.article_id.astype('int32')\ntransactions.t_dat = cudf.to_datetime(transactions.t_dat)\ntransactions = transactions[['t_dat','customer_id','article_id']]\n#transactions.to_parquet('train.pqt',index=False)\nprint( transactions.shape )\ntransactions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.169556Z","iopub.status.idle":"2025-01-15T15:01:15.169794Z","shell.execute_reply":"2025-01-15T15:01:15.169697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Grouper les données et calculer le nombre d'achats\ntmp = transactions.groupby(['customer_id','article_id'])['t_dat'].agg('count').reset_index()\ntmp.columns = ['customer_id','article_id','ct']\ntmp.tail() #Afficher les dernières lignes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.17039Z","iopub.status.idle":"2025-01-15T15:01:15.170686Z","shell.execute_reply":"2025-01-15T15:01:15.170573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions = transactions.merge(tmp,on=['customer_id','article_id'],how='left') #Fusionner les données avec tmp\ntransactions = transactions.sort_values(['ct','t_dat'],ascending=False)#Trier les transactions\ntransactions = transactions.drop_duplicates(['customer_id','article_id'])#Supprimer les doublons\ntransactions = transactions.sort_values(['ct','t_dat'],ascending=False)#Trier à nouveau les transactions\ntransactions.tail()#Afficher les dernières lignes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.171371Z","iopub.status.idle":"2025-01-15T15:01:15.171685Z","shell.execute_reply":"2025-01-15T15:01:15.171569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions['year'] = transactions['t_dat'].dt.year #Extraire l'année de t_dat\ntransactions['month'] = transactions['t_dat'].dt.month #\ntransactions['day'] = transactions['t_dat'].dt.day\ntransactions['dayofweek'] = transactions['t_dat'].dt.dayofweek\ntransactions['dayofyear'] = transactions['t_dat'].dt.dayofyear\ntransactions['is_month_end'] = transactions['t_dat'].dt.is_month_end\ntransactions['is_month_start'] = transactions['t_dat'].dt.is_month_start\ntransactions.drop(columns=['t_dat'], inplace = True)\n\ntransactions.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.172594Z","iopub.status.idle":"2025-01-15T15:01:15.172924Z","shell.execute_reply":"2025-01-15T15:01:15.172811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions['cust_cat']= transactions['customer_id'].astype('category')\ntransactions['cat_codes'] = transactions['cust_cat'].cat.codes \ncust_cat_df = transactions[['customer_id', 'cust_cat', 'cat_codes']] #save them to put them back together later\nprint(cust_cat_df.dtypes)\ncust_cat_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.173634Z","iopub.status.idle":"2025-01-15T15:01:15.173868Z","shell.execute_reply":"2025-01-15T15:01:15.173771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions.drop(columns=['cust_cat', 'cat_codes'], inplace = True)\ntransactions.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.174408Z","iopub.status.idle":"2025-01-15T15:01:15.174661Z","shell.execute_reply":"2025-01-15T15:01:15.174564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Put number sold into articles DF\narticles = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\narticles.drop(columns=['detail_desc'], inplace = True)\narticles.shape, articles.dtypes\narticles","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.17513Z","iopub.status.idle":"2025-01-15T15:01:15.175408Z","shell.execute_reply":"2025-01-15T15:01:15.175266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_names= articles.select_dtypes(include=['object']).columns\ncont_names = articles.select_dtypes(include=['int64']).columns\nobj_names = articles.select_dtypes(include=['object']).columns\n\nfor i in cat_names: articles[i+'_cat']=articles[i].astype('category')\nfor i in obj_names: articles.drop(columns=[i], inplace = True)\n\narticles.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.176085Z","iopub.status.idle":"2025-01-15T15:01:15.176356Z","shell.execute_reply":"2025-01-15T15:01:15.176244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.177251Z","iopub.status.idle":"2025-01-15T15:01:15.177661Z","shell.execute_reply":"2025-01-15T15:01:15.177459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"times_bought = transactions[['article_id', 'ct']]\ntimes_bought = times_bought.groupby('article_id', as_index = False).sum()\ntimes_bought.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.178689Z","iopub.status.idle":"2025-01-15T15:01:15.179094Z","shell.execute_reply":"2025-01-15T15:01:15.178897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles = articles.merge(times_bought,  how='left', on='article_id')\narticles['ct'] = articles['ct'].fillna(0)\narticles.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.179778Z","iopub.status.idle":"2025-01-15T15:01:15.180079Z","shell.execute_reply":"2025-01-15T15:01:15.179942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_names= articles.select_dtypes(include=['category']).columns\narticle_cat_df = cudf.DataFrame()\n\nfor i in cat_names: \n    articles[i+'_cat_code'] = articles[i].cat.codes\n    \n    #save them to put them back together later\n    article_cat_df[i] = articles[i]    \n    article_cat_df[i+'cat_code'] =articles[i+'_cat_code']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.180708Z","iopub.status.idle":"2025-01-15T15:01:15.180983Z","shell.execute_reply":"2025-01-15T15:01:15.180866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"article_cat_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.181783Z","iopub.status.idle":"2025-01-15T15:01:15.182133Z","shell.execute_reply":"2025-01-15T15:01:15.181947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#something with cudf that this needs to be in a loop, very fast anyway\nfor i in cat_names:\n    articles.drop(columns=[i], inplace = True)\narticles.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.182883Z","iopub.status.idle":"2025-01-15T15:01:15.183276Z","shell.execute_reply":"2025-01-15T15:01:15.183077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#kmeans can't handle integers, so convert to float\nint64s = articles.select_dtypes(include=['int64']).columns\nfor i in int64s:\n    articles[i] = articles[i].astype(float)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.184081Z","iopub.status.idle":"2025-01-15T15:01:15.184376Z","shell.execute_reply":"2025-01-15T15:01:15.184249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Simple random forest to predict # of sales\ncols_list = articles.columns\ncols_list = cols_list.to_list()\ncols_list.remove('ct')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.18507Z","iopub.status.idle":"2025-01-15T15:01:15.185375Z","shell.execute_reply":"2025-01-15T15:01:15.185241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# convert to pandas, use scikit learn random forest to see what features are useful\n# unfortunately cuML doesn't have feature importance in their random forest yet...\nX = articles[cols_list].to_pandas()\ny = articles['ct'].to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.186047Z","iopub.status.idle":"2025-01-15T15:01:15.186349Z","shell.execute_reply":"2025-01-15T15:01:15.18621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.datasets import make_regression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom matplotlib import pyplot\n# define dataset\n#X, y = make_regression(n_samples=1000, n_features=10, n_informative=5, random_state=1)\n# define the model\n#model = DecisionTreeRegressor()\nmodel = RandomForestRegressor()\n# fit the model\nmodel.fit(X, y)\n# get importance","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.187067Z","iopub.status.idle":"2025-01-15T15:01:15.187407Z","shell.execute_reply":"2025-01-15T15:01:15.187226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fi_plot_df = pd.DataFrame({'cols':X.columns, 'imp':model.feature_importances_}).sort_values('imp', ascending=False)    \nfi_plot_df.plot(kind=\"barh\", x = 'cols')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.18824Z","iopub.status.idle":"2025-01-15T15:01:15.188575Z","shell.execute_reply":"2025-01-15T15:01:15.188419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# so from above, these are the columns that are important to pass to k-means\narticles = articles[['article_id', 'prod_name_cat_cat_code', 'product_code', 'department_no', 'colour_group_name_cat_cat_code', 'ct']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.189413Z","iopub.status.idle":"2025-01-15T15:01:15.191871Z","shell.execute_reply":"2025-01-15T15:01:15.189594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install scikit-learn matplotlib","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.19273Z","iopub.status.idle":"2025-01-15T15:01:15.193066Z","shell.execute_reply":"2025-01-15T15:01:15.192941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Using K-means to create a feature on articles\n\n\nUsing K-means to create a feature on articles\nSum_of_squared_distances = []\nK = range(1, 10)\nfor num_clusters in K :\n kmeans = KMeans(n_clusters=num_clusters)\n kmeans.fit(articles)\n Sum_of_squared_distances.append(kmeans.inertia_)\nplt.plot(K,Sum_of_squared_distances,'bx-')\nplt.xlabel('Values of K') \nplt.ylabel('Sum of squared distances/Inertia') \nplt.title('Elbow Method For Optimal k')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.193818Z","iopub.status.idle":"2025-01-15T15:01:15.194105Z","shell.execute_reply":"2025-01-15T15:01:15.19399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# elbow looks like around 4\n# It was 12 in an an earlier version!\nkmeans_float = KMeans(n_clusters=4)\nkmeans_fit = kmeans_float.fit(articles)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.194921Z","iopub.status.idle":"2025-01-15T15:01:15.195274Z","shell.execute_reply":"2025-01-15T15:01:15.195088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"labels:\")\nprint(kmeans_float.labels_)\nprint(\"cluster_centers:\")\nprint(kmeans_float.cluster_centers_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.196327Z","iopub.status.idle":"2025-01-15T15:01:15.196672Z","shell.execute_reply":"2025-01-15T15:01:15.196543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kmeans_float.fit_predict(articles)\nlabels = kmeans_float.labels_\n\n#Glue back to originaal data\narticles['clusters'] = labels\narticles.clusters.value_counts()\narticles.tail()\narticles.to_parquet('articles.parquet', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-15T15:01:15.197279Z","iopub.status.idle":"2025-01-15T15:01:15.19761Z","shell.execute_reply":"2025-01-15T15:01:15.197448Z"}},"outputs":[],"execution_count":null}]}