{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-28T12:12:04.245932Z","iopub.execute_input":"2026-01-28T12:12:04.246232Z","iopub.status.idle":"2026-01-28T12:16:46.066157Z","shell.execute_reply.started":"2026-01-28T12:12:04.246191Z","shell.execute_reply":"2026-01-28T12:16:46.065043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\npath = '/kaggle/input/h-and-m-personalized-fashion-recommendations/'\n\narticles = pd.read_csv(path + 'articles.csv')\ncustomers = pd.read_csv(path + 'customers.csv')\ntransactions = pd.read_csv(path + 'transactions_train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:27:00.867355Z","iopub.execute_input":"2026-01-28T17:27:00.868009Z","iopub.status.idle":"2026-01-28T17:28:10.870977Z","shell.execute_reply.started":"2026-01-28T17:27:00.867975Z","shell.execute_reply":"2026-01-28T17:28:10.869882Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing du dataset **Articles**","metadata":{}},{"cell_type":"code","source":"articles.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:28:21.173049Z","iopub.execute_input":"2026-01-28T17:28:21.173997Z","iopub.status.idle":"2026-01-28T17:28:21.271979Z","shell.execute_reply.started":"2026-01-28T17:28:21.173963Z","shell.execute_reply":"2026-01-28T17:28:21.271029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:28:25.735726Z","iopub.execute_input":"2026-01-28T17:28:25.736053Z","iopub.status.idle":"2026-01-28T17:28:25.75679Z","shell.execute_reply.started":"2026-01-28T17:28:25.736025Z","shell.execute_reply":"2026-01-28T17:28:25.755581Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Gestion des valeurs manquantes","metadata":{}},{"cell_type":"markdown","source":"D’après notre résumé : UNE seule colonne contient des valeurs manquantes.\n\n    detail_desc : 105126 non-null sur 105542\n    ≈ 416 valeurs manquantes (~0.4%)\n    \ndetail_desc donne une description textuelle du produit.\nce qui n'est pas necessaire pour clustring.","metadata":{}},{"cell_type":"code","source":"articles = articles.drop(columns=['detail_desc'])\narticles.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:28:32.646689Z","iopub.execute_input":"2026-01-28T17:28:32.647092Z","iopub.status.idle":"2026-01-28T17:28:32.693168Z","shell.execute_reply.started":"2026-01-28T17:28:32.647056Z","shell.execute_reply":"2026-01-28T17:28:32.692143Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##### La variable textuelle n’étant pas exploitée dans notre modèle, elle a été retirée pour réduire la dimensionalité.","metadata":{}},{"cell_type":"markdown","source":"## Gestion des redondances","metadata":{}},{"cell_type":"markdown","source":"Le type de redondance présent : Même information, deux formats différents\n( ID numérique + nom textuel).\n\nOn garde les colonnes textuelles, on supprime les codes numériques.","metadata":{}},{"cell_type":"code","source":"cols_to_drop = [\n    'product_code',\n    'product_type_no',\n    'graphical_appearance_no',\n    'colour_group_code',\n    'perceived_colour_value_id',\n    'perceived_colour_master_id',\n    'department_no',\n    'index_group_no',\n    'section_no',\n    'garment_group_no'\n]\n\narticles_clean = articles.drop(columns=cols_to_drop)\narticles_clean.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:28:41.116146Z","iopub.execute_input":"2026-01-28T17:28:41.116522Z","iopub.status.idle":"2026-01-28T17:28:41.150716Z","shell.execute_reply.started":"2026-01-28T17:28:41.116489Z","shell.execute_reply":"2026-01-28T17:28:41.149663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Verification de l'unicite de l'ID d'article\narticles_clean['article_id'].duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:28:48.633974Z","iopub.execute_input":"2026-01-28T17:28:48.634309Z","iopub.status.idle":"2026-01-28T17:28:48.646029Z","shell.execute_reply.started":"2026-01-28T17:28:48.634281Z","shell.execute_reply":"2026-01-28T17:28:48.644981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verification des lignes identiques à 100 %\narticles_clean.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:28:49.905577Z","iopub.execute_input":"2026-01-28T17:28:49.906723Z","iopub.status.idle":"2026-01-28T17:28:50.027269Z","shell.execute_reply.started":"2026-01-28T17:28:49.906681Z","shell.execute_reply":"2026-01-28T17:28:50.026228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles_final = articles_clean[[\n    'article_id',\n    'product_type_name',\n    'product_group_name',\n    'graphical_appearance_name',\n    'colour_group_name',\n    'perceived_colour_value_name',\n    'perceived_colour_master_name',\n    'department_name',\n    'index_name',\n    'index_group_name',\n    'section_name',\n    'garment_group_name'\n]]\narticles_final.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:28:53.57834Z","iopub.execute_input":"2026-01-28T17:28:53.579325Z","iopub.status.idle":"2026-01-28T17:28:54.113562Z","shell.execute_reply.started":"2026-01-28T17:28:53.57929Z","shell.execute_reply":"2026-01-28T17:28:54.112223Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Une analyse exploratoire a permis de vérifier le nombre de catégories par variable ainsi que l’absence de doublons logiques. L’identifiant article_id s’est avéré unique, confirmant l’intégrité du jeu de données.","metadata":{}},{"cell_type":"code","source":"summary = pd.DataFrame({\n    'nb_valeurs_uniques': articles_final.select_dtypes(include='object').nunique(),\n    'nb_valeurs_manquantes': articles_final.select_dtypes(include='object').isnull().sum()\n})\n\nsummary\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:28:57.79641Z","iopub.execute_input":"2026-01-28T17:28:57.797582Z","iopub.status.idle":"2026-01-28T17:28:57.960162Z","shell.execute_reply.started":"2026-01-28T17:28:57.797543Z","shell.execute_reply":"2026-01-28T17:28:57.959149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\narticles_df = articles_final.copy()\n\nfor col in articles_df.columns:\n    if col !='article_id':\n        le = LabelEncoder()\n        articles_df[col] = le.fit_transform(\n            articles_df[col].astype(str)\n        )\narticles_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:29:00.426402Z","iopub.execute_input":"2026-01-28T17:29:00.42708Z","iopub.status.idle":"2026-01-28T17:29:00.661732Z","shell.execute_reply.started":"2026-01-28T17:29:00.427047Z","shell.execute_reply":"2026-01-28T17:29:00.660485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:29:06.155044Z","iopub.execute_input":"2026-01-28T17:29:06.155457Z","iopub.status.idle":"2026-01-28T17:29:06.168757Z","shell.execute_reply.started":"2026-01-28T17:29:06.155424Z","shell.execute_reply":"2026-01-28T17:29:06.167767Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# Preprocessing du dataset **Customers**","metadata":{}},{"cell_type":"code","source":"customers.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:06:51.370334Z","iopub.execute_input":"2026-01-28T17:06:51.371231Z","iopub.status.idle":"2026-01-28T17:06:51.731029Z","shell.execute_reply.started":"2026-01-28T17:06:51.371191Z","shell.execute_reply":"2026-01-28T17:06:51.730177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"customers.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:06:52.970542Z","iopub.execute_input":"2026-01-28T17:06:52.97088Z","iopub.status.idle":"2026-01-28T17:06:52.983811Z","shell.execute_reply.started":"2026-01-28T17:06:52.970854Z","shell.execute_reply":"2026-01-28T17:06:52.982675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"summary_1 = pd.DataFrame({\n    'nb_valeurs_uniques': customers.nunique(),\n    'valeurs_exemple': [customers[col].unique()[:10] for col in customers.columns]\n})\n\nsummary_1\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:06:57.306327Z","iopub.execute_input":"2026-01-28T17:06:57.306702Z","iopub.status.idle":"2026-01-28T17:06:59.719465Z","shell.execute_reply.started":"2026-01-28T17:06:57.306673Z","shell.execute_reply":"2026-01-28T17:06:59.718176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\ncustomers_df = customers.copy()\n\n# Remplacer NaN pour flags\ncustomers_df['FN'] = customers_df['FN'].fillna(0)\ncustomers_df['Active'] = customers_df['Active'].fillna(0)\n\n# Remplacer NaN pour catégoriel\ncustomers_df['club_member_status'] = customers_df['club_member_status'].fillna('unknown')\ncustomers_df['fashion_news_frequency'] = customers_df['fashion_news_frequency'].fillna('unknown')\n\n# Remplacer NaN pour âge\ncustomers_df['age'] = customers_df['age'].fillna(customers_df['age'].median())\n\n# Encodage des colonnes catégorielles \ncat_cols = ['club_member_status', 'fashion_news_frequency']\nfor col in cat_cols:\n    le = LabelEncoder()\n    customers_df[col] = le.fit_transform(customers_df[col].astype(str))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:06:59.721364Z","iopub.execute_input":"2026-01-28T17:06:59.721755Z","iopub.status.idle":"2026-01-28T17:07:00.554739Z","shell.execute_reply.started":"2026-01-28T17:06:59.721726Z","shell.execute_reply":"2026-01-28T17:07:00.553698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"customers_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:07:00.556527Z","iopub.execute_input":"2026-01-28T17:07:00.556921Z","iopub.status.idle":"2026-01-28T17:07:00.74302Z","shell.execute_reply.started":"2026-01-28T17:07:00.556881Z","shell.execute_reply":"2026-01-28T17:07:00.742042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"customers_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:07:01.246904Z","iopub.execute_input":"2026-01-28T17:07:01.247679Z","iopub.status.idle":"2026-01-28T17:07:01.260836Z","shell.execute_reply.started":"2026-01-28T17:07:01.247645Z","shell.execute_reply":"2026-01-28T17:07:01.259523Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing du dataset **Transactions**","metadata":{}},{"cell_type":"code","source":"transactions.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:07:06.159473Z","iopub.execute_input":"2026-01-28T17:07:06.16044Z","iopub.status.idle":"2026-01-28T17:07:06.16984Z","shell.execute_reply.started":"2026-01-28T17:07:06.160405Z","shell.execute_reply":"2026-01-28T17:07:06.16844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:07:07.4739Z","iopub.execute_input":"2026-01-28T17:07:07.474858Z","iopub.status.idle":"2026-01-28T17:07:07.484519Z","shell.execute_reply.started":"2026-01-28T17:07:07.47482Z","shell.execute_reply":"2026-01-28T17:07:07.483763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transaction_copie = transactions.copy()\ntransaction_copie['t_dat'] = pd.to_datetime(transaction_copie['t_dat'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:07:10.996037Z","iopub.execute_input":"2026-01-28T17:07:10.997189Z","iopub.status.idle":"2026-01-28T17:07:15.800575Z","shell.execute_reply.started":"2026-01-28T17:07:10.997155Z","shell.execute_reply":"2026-01-28T17:07:15.799457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transaction_copie.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:07:15.802251Z","iopub.execute_input":"2026-01-28T17:07:15.802675Z","iopub.status.idle":"2026-01-28T17:07:15.811074Z","shell.execute_reply.started":"2026-01-28T17:07:15.802632Z","shell.execute_reply":"2026-01-28T17:07:15.809942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transaction_copie['t_dat'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:07:15.812365Z","iopub.execute_input":"2026-01-28T17:07:15.812875Z","iopub.status.idle":"2026-01-28T17:07:15.829821Z","shell.execute_reply.started":"2026-01-28T17:07:15.812835Z","shell.execute_reply":"2026-01-28T17:07:15.828893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Vérifier les valeurs manquantes\ntransaction_copie.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:07:15.935673Z","iopub.execute_input":"2026-01-28T17:07:15.936006Z","iopub.status.idle":"2026-01-28T17:07:17.80572Z","shell.execute_reply.started":"2026-01-28T17:07:15.93598Z","shell.execute_reply":"2026-01-28T17:07:17.804497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Vérifier les doublons\ntransaction_copie.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:07:17.926522Z","iopub.execute_input":"2026-01-28T17:07:17.926877Z","iopub.status.idle":"2026-01-28T17:07:36.530826Z","shell.execute_reply.started":"2026-01-28T17:07:17.926849Z","shell.execute_reply":"2026-01-28T17:07:36.529936Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Les doublons exacts doit etre supprimer\n\nSi tu les gardes → tu comptes plusieurs fois la même transaction → biais dans les fréquences de vente.\n\nSi tu les supprimes → tu gardes une transaction par vente réelle → plus propre pour le clustering ou collaborative filtering.","metadata":{}},{"cell_type":"code","source":"transaction_copie = transaction_copie.drop_duplicates()\ntransaction_copie.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:08:15.803549Z","iopub.execute_input":"2026-01-28T17:08:15.803905Z","iopub.status.idle":"2026-01-28T17:08:53.500738Z","shell.execute_reply.started":"2026-01-28T17:08:15.803877Z","shell.execute_reply":"2026-01-28T17:08:53.499733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combinaison unique client + article + date ?\ntransaction_copie[['t_dat', 'customer_id', 'article_id']].duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:08:56.4323Z","iopub.execute_input":"2026-01-28T17:08:56.432613Z","iopub.status.idle":"2026-01-28T17:09:13.865338Z","shell.execute_reply.started":"2026-01-28T17:08:56.432569Z","shell.execute_reply":"2026-01-28T17:09:13.864364Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Les doublons exacts sur la combinaison t_dat, customer_id et article_id ont été supprimés (~0,75 % des transactions) afin de garantir l’unicité des ventes et d’éviter de biaiser la matrice client-produit.","metadata":{}},{"cell_type":"code","source":"transaction_copie = transaction_copie.drop_duplicates(subset=['t_dat', 'customer_id', 'article_id'])\ntransaction_copie[['t_dat', 'customer_id', 'article_id']].duplicated().sum()  # → doit être 0\ntransaction_copie.shape  # pour voir combien de lignes restent","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:09:18.26036Z","iopub.execute_input":"2026-01-28T17:09:18.260739Z","iopub.status.idle":"2026-01-28T17:09:53.950688Z","shell.execute_reply.started":"2026-01-28T17:09:18.260709Z","shell.execute_reply":"2026-01-28T17:09:53.949462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transaction_copie = transaction_copie.reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:09:58.255258Z","iopub.execute_input":"2026-01-28T17:09:58.255681Z","iopub.status.idle":"2026-01-28T17:09:59.097827Z","shell.execute_reply.started":"2026-01-28T17:09:58.255652Z","shell.execute_reply":"2026-01-28T17:09:59.096981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transaction_copie['price'].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:09:59.647335Z","iopub.execute_input":"2026-01-28T17:09:59.647884Z","iopub.status.idle":"2026-01-28T17:10:00.607279Z","shell.execute_reply.started":"2026-01-28T17:09:59.647852Z","shell.execute_reply":"2026-01-28T17:10:00.606313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\ntransaction_copie['price_scaled'] = scaler.fit_transform(transaction_copie[['price']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:10:00.721648Z","iopub.execute_input":"2026-01-28T17:10:00.722476Z","iopub.status.idle":"2026-01-28T17:10:01.31599Z","shell.execute_reply.started":"2026-01-28T17:10:00.722443Z","shell.execute_reply":"2026-01-28T17:10:01.315116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transaction_copie.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:10:03.017184Z","iopub.execute_input":"2026-01-28T17:10:03.017713Z","iopub.status.idle":"2026-01-28T17:10:03.028356Z","shell.execute_reply.started":"2026-01-28T17:10:03.017671Z","shell.execute_reply":"2026-01-28T17:10:03.026868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transaction_copie.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:10:04.839099Z","iopub.execute_input":"2026-01-28T17:10:04.839865Z","iopub.status.idle":"2026-01-28T17:10:04.853863Z","shell.execute_reply.started":"2026-01-28T17:10:04.839828Z","shell.execute_reply":"2026-01-28T17:10:04.852642Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fusion des datasets","metadata":{}},{"cell_type":"code","source":"trans_cust_df = transaction_copie.merge(customers_df, on='customer_id', how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:10:08.973395Z","iopub.execute_input":"2026-01-28T17:10:08.974444Z","iopub.status.idle":"2026-01-28T17:10:26.121102Z","shell.execute_reply.started":"2026-01-28T17:10:08.974383Z","shell.execute_reply":"2026-01-28T17:10:26.120006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df = trans_cust_df.merge(articles_df, on='article_id', how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:10:26.122738Z","iopub.execute_input":"2026-01-28T17:10:26.123112Z","iopub.status.idle":"2026-01-28T17:10:41.33682Z","shell.execute_reply.started":"2026-01-28T17:10:26.123086Z","shell.execute_reply":"2026-01-28T17:10:41.336022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T16:22:41.542424Z","iopub.execute_input":"2026-01-28T16:22:41.542835Z","iopub.status.idle":"2026-01-28T16:22:41.564055Z","shell.execute_reply.started":"2026-01-28T16:22:41.542805Z","shell.execute_reply":"2026-01-28T16:22:41.563243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:11:08.471424Z","iopub.execute_input":"2026-01-28T17:11:08.472199Z","iopub.status.idle":"2026-01-28T17:11:08.483705Z","shell.execute_reply.started":"2026-01-28T17:11:08.472164Z","shell.execute_reply":"2026-01-28T17:11:08.482396Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature engeneering","metadata":{}},{"cell_type":"code","source":"# Nombre total d'achats par client\nclient_freq = final_df.groupby('customer_id').size().rename('total_purchases')\nfinal_df = final_df.merge(client_freq, on='customer_id', how='left')\n\n# Nombre d'articles uniques achetés\nclient_unique_articles = final_df.groupby('customer_id')['article_id'].nunique().rename('unique_articles')\nfinal_df = final_df.merge(client_unique_articles, on='customer_id', how='left')\n\n# Total dépensé par client\nclient_total_spent = final_df.groupby('customer_id')['price'].sum().rename('total_spent')\nfinal_df = final_df.merge(client_total_spent, on='customer_id', how='left')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:12:29.36518Z","iopub.execute_input":"2026-01-28T17:12:29.366163Z","iopub.status.idle":"2026-01-28T17:14:00.88629Z","shell.execute_reply.started":"2026-01-28T17:12:29.366129Z","shell.execute_reply":"2026-01-28T17:14:00.885126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Nombre de clients différents\narticle_popularity = final_df.groupby('article_id')['customer_id'].nunique().rename('num_customers')\nfinal_df = final_df.merge(article_popularity, on='article_id', how='left')\n\n# Nombre total de ventes\narticle_sales = final_df.groupby('article_id').size().rename('total_sales')\nfinal_df = final_df.merge(article_sales, on='article_id', how='left')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:14:25.803007Z","iopub.execute_input":"2026-01-28T17:14:25.804125Z","iopub.status.idle":"2026-01-28T17:15:03.791017Z","shell.execute_reply.started":"2026-01-28T17:14:25.804089Z","shell.execute_reply":"2026-01-28T17:15:03.790024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df['year'] = final_df['t_dat'].dt.year\nfinal_df['month'] = final_df['t_dat'].dt.month\nfinal_df['weekday'] = final_df['t_dat'].dt.weekday","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:15:31.871441Z","iopub.execute_input":"2026-01-28T17:15:31.872294Z","iopub.status.idle":"2026-01-28T17:15:34.507471Z","shell.execute_reply.started":"2026-01-28T17:15:31.872246Z","shell.execute_reply":"2026-01-28T17:15:34.505982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Liste finale des colonnes\ncols_ordered = [\n    'customer_id','article_id','t_dat','sales_channel_id','price','price_scaled',\n    'age','FN','Active','club_member_status','fashion_news_frequency','postal_code',\n    'product_type_name','product_group_name','graphical_appearance_name',\n    'colour_group_name','perceived_colour_value_name','perceived_colour_master_name',\n    'department_name','index_name','index_group_name','section_name',\n    'garment_group_name',\n    'year','month','weekday',\n    'total_purchases','unique_articles','total_spent',\n    'num_customers','total_sales'\n]\n\nfinal_df = final_df[cols_ordered]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:29:56.420228Z","iopub.execute_input":"2026-01-28T17:29:56.421091Z","iopub.status.idle":"2026-01-28T17:30:05.349095Z","shell.execute_reply.started":"2026-01-28T17:29:56.421051Z","shell.execute_reply":"2026-01-28T17:30:05.348012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:30:26.67966Z","iopub.execute_input":"2026-01-28T17:30:26.680498Z","iopub.status.idle":"2026-01-28T17:30:26.701496Z","shell.execute_reply.started":"2026-01-28T17:30:26.680438Z","shell.execute_reply":"2026-01-28T17:30:26.700209Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Construction du model","metadata":{}},{"cell_type":"markdown","source":"Logique globale :\n\nRegrouper les articles similaires (couleur, type, groupe, prix, etc.)\nPuis recommander des produits du même cluster.\n\n1. final_df → contient des transactions\n\n2. On doit créer un dataset article-level (1 ligne = 1 article)\n\n3. On encode + normalise les features\n\n4. PCA → réduire la dimension\n\n5. K-Means → créer des clusters d’articles\n\n6. Recommandation = articles du même cluster","metadata":{}},{"cell_type":"code","source":"article_features = final_df.groupby('article_id').agg({\n    'price': 'mean',\n    'product_type_name': 'first',\n    'product_group_name': 'first',\n    'graphical_appearance_name': 'first',\n    'colour_group_name': 'first',\n    'perceived_colour_master_name': 'first',\n    'department_name': 'first',\n    'index_group_name': 'first',\n    'garment_group_name': 'first',\n    'total_sales': 'first',      \n    'num_customers': 'first'\n}).reset_index()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T17:57:52.913704Z","iopub.execute_input":"2026-01-28T17:57:52.914811Z","iopub.status.idle":"2026-01-28T17:57:58.217647Z","shell.execute_reply.started":"2026-01-28T17:57:52.914775Z","shell.execute_reply":"2026-01-28T17:57:58.216689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\n\nX = article_features.drop(columns=['article_id'])\nX_scaled = scaler.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T18:00:42.924221Z","iopub.execute_input":"2026-01-28T18:00:42.925057Z","iopub.status.idle":"2026-01-28T18:00:42.956347Z","shell.execute_reply.started":"2026-01-28T18:00:42.925024Z","shell.execute_reply":"2026-01-28T18:00:42.95517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\npca = PCA(n_components=0.90)  # garder 90% de l'information\nX_pca = pca.fit_transform(X_scaled)\n\nprint(\"Nombre de composantes PCA :\", X_pca.shape[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T18:00:57.481212Z","iopub.execute_input":"2026-01-28T18:00:57.481533Z","iopub.status.idle":"2026-01-28T18:00:57.879387Z","shell.execute_reply.started":"2026-01-28T18:00:57.481506Z","shell.execute_reply":"2026-01-28T18:00:57.877676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.cluster import KMeans\n\nkmeans = KMeans(\n    n_clusters=30,\n    random_state=42,\n    n_init=10\n)\n\narticle_features['cluster'] = kmeans.fit_predict(X_pca)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T18:01:16.298838Z","iopub.execute_input":"2026-01-28T18:01:16.299333Z","iopub.status.idle":"2026-01-28T18:01:22.093495Z","shell.execute_reply.started":"2026-01-28T18:01:16.299306Z","shell.execute_reply":"2026-01-28T18:01:22.092418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def recommend_similar_products(article_id, n=5):\n    cluster_id = article_features.loc[\n        article_features['article_id'] == article_id, 'cluster'\n    ].values[0]\n\n    recommendations = article_features[\n        article_features['cluster'] == cluster_id\n    ].sample(n)\n\n    return recommendations[['article_id', 'cluster']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T18:01:56.877397Z","iopub.execute_input":"2026-01-28T18:01:56.877946Z","iopub.status.idle":"2026-01-28T18:01:56.884762Z","shell.execute_reply.started":"2026-01-28T18:01:56.877892Z","shell.execute_reply":"2026-01-28T18:01:56.883788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"recommend_similar_products(108775044, n=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T18:02:07.111656Z","iopub.execute_input":"2026-01-28T18:02:07.112744Z","iopub.status.idle":"2026-01-28T18:02:07.127977Z","shell.execute_reply.started":"2026-01-28T18:02:07.112709Z","shell.execute_reply":"2026-01-28T18:02:07.126385Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Recommander à un client des articles qu’il n’a jamais achetés,\n\nmais qui sont similaires à ceux qu’il a déjà achetés,\n\nen se basant sur les clusters d’articles.\n\n  Entrée : customer_id\n  \n  Sortie : liste de article_id recommandés","metadata":{}},{"cell_type":"code","source":"final_DF=final_df.copy()\nfinal_DF = final_DF.merge(\n    article_features[['article_id', 'cluster']],\n    on='article_id',\n    how='left'\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T18:08:27.142954Z","iopub.execute_input":"2026-01-28T18:08:27.144368Z","execution_failed":"2026-01-28T18:08:56.656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_DF.head","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}