{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":30527,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# FASHION RECOMMENDER SYSTEM / CONTENT-BASED FILTERING","metadata":{}},{"cell_type":"markdown","source":"## 1. Importing Libraries and Data","metadata":{}},{"cell_type":"code","source":"# Importing libraries\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport re\nfrom nltk.corpus import stopwords\nfrom nltk.stem import PorterStemmer\nfrom nltk.tokenize import word_tokenize\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics.pairwise import cosine_similarity\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport random\n\n# # Set display options\npd.set_option('display.max_columns', None)\npd.set_option('display.width', 500)\npd.set_option('display.expand_frame_repr', False)\nimport warnings\nwarnings.simplefilter(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:50:04.45356Z","iopub.execute_input":"2023-09-30T11:50:04.454002Z","iopub.status.idle":"2023-09-30T11:50:07.319652Z","shell.execute_reply.started":"2023-09-30T11:50:04.453968Z","shell.execute_reply":"2023-09-30T11:50:07.318496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the CSV file\n\ndf_articles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ndf_articles.head()\ndf_articles.shape\n\n#(105542, 25)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:50:07.322155Z","iopub.execute_input":"2023-09-30T11:50:07.322976Z","iopub.status.idle":"2023-09-30T11:50:08.772161Z","shell.execute_reply.started":"2023-09-30T11:50:07.322931Z","shell.execute_reply":"2023-09-30T11:50:08.77103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Selecting the ones with 'ladies wear' from the 'index_group_name'\n\ndf_ladies = df_articles[df_articles['index_group_name'] == 'Ladieswear']\ndf_ladies = df_ladies.reset_index(drop=True)\ndf_ladies.shape\ndf_ladies.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:50:08.773691Z","iopub.execute_input":"2023-09-30T11:50:08.774009Z","iopub.status.idle":"2023-09-30T11:50:08.836589Z","shell.execute_reply.started":"2023-09-30T11:50:08.773981Z","shell.execute_reply":"2023-09-30T11:50:08.835553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Data Preprocessing","metadata":{}},{"cell_type":"code","source":"# Identifying missing values\n\ndf_ladies.isnull().any()\nmissing_count = df_ladies['detail_desc'].isnull().sum()\nprint(missing_count)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:50:08.8394Z","iopub.execute_input":"2023-09-30T11:50:08.839755Z","iopub.status.idle":"2023-09-30T11:50:08.974448Z","shell.execute_reply.started":"2023-09-30T11:50:08.839728Z","shell.execute_reply":"2023-09-30T11:50:08.973067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Removing missing data\n\ndf_ladies = df_ladies.dropna(subset=['detail_desc'])\ndf_ladies = df_ladies.reset_index(drop=True)\ndf_ladies.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:50:08.97581Z","iopub.execute_input":"2023-09-30T11:50:08.976214Z","iopub.status.idle":"2023-09-30T11:50:09.024918Z","shell.execute_reply.started":"2023-09-30T11:50:08.976185Z","shell.execute_reply":"2023-09-30T11:50:09.023689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Total number of products in each product group\n\ndf_ladies['product_group_name'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:50:09.026372Z","iopub.execute_input":"2023-09-30T11:50:09.027092Z","iopub.status.idle":"2023-09-30T11:50:09.043996Z","shell.execute_reply.started":"2023-09-30T11:50:09.027035Z","shell.execute_reply":"2023-09-30T11:50:09.043011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Determining the data type of columns\n\ndf_ladies.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:50:09.045269Z","iopub.execute_input":"2023-09-30T11:50:09.046378Z","iopub.status.idle":"2023-09-30T11:50:09.059343Z","shell.execute_reply.started":"2023-09-30T11:50:09.046145Z","shell.execute_reply":"2023-09-30T11:50:09.058378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preparation for TF-IDF matrix\n\ndf = df_ladies.select_dtypes(include=['object'])\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:50:09.060734Z","iopub.execute_input":"2023-09-30T11:50:09.061443Z","iopub.status.idle":"2023-09-30T11:50:09.088709Z","shell.execute_reply.started":"2023-09-30T11:50:09.061403Z","shell.execute_reply":"2023-09-30T11:50:09.087348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate all text columns to create a document collection\n\ndocuments = df.apply(' '.join, axis=1)\nprint(documents)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:50:09.090117Z","iopub.execute_input":"2023-09-30T11:50:09.090593Z","iopub.status.idle":"2023-09-30T11:50:09.34483Z","shell.execute_reply.started":"2023-09-30T11:50:09.090554Z","shell.execute_reply":"2023-09-30T11:50:09.34347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocessing for TF-IDF vectors\n\ndef preprocess_text(text):\n    text = text.lower()\n    text = re.sub(r'[^\\w\\s]', ' ', text)\n    tokens = text.split()\n    filtered_tokens = [word for word in tokens if word not in stopwords.words('english')]\n    return \" \".join(filtered_tokens)\n\ndocuments = documents.apply(preprocess_text)\nprint(documents)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:50:09.34605Z","iopub.execute_input":"2023-09-30T11:50:09.346409Z","iopub.status.idle":"2023-09-30T11:53:21.624406Z","shell.execute_reply.started":"2023-09-30T11:50:09.346351Z","shell.execute_reply":"2023-09-30T11:53:21.623302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. TF-IDF Vectorization","metadata":{}},{"cell_type":"code","source":"# TF-IDF vectorization\n\ntfidf = TfidfVectorizer()","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:53:21.627191Z","iopub.execute_input":"2023-09-30T11:53:21.627811Z","iopub.status.idle":"2023-09-30T11:53:21.63222Z","shell.execute_reply.started":"2023-09-30T11:53:21.627781Z","shell.execute_reply":"2023-09-30T11:53:21.631376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert the document collection to TF-IDF vectors\n\ntfidf_matrix = tfidf.fit_transform(documents)\nprint(tfidf_matrix.toarray())","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:53:21.633199Z","iopub.execute_input":"2023-09-30T11:53:21.633494Z","iopub.status.idle":"2023-09-30T11:53:27.39024Z","shell.execute_reply.started":"2023-09-30T11:53:21.633469Z","shell.execute_reply":"2023-09-30T11:53:27.389027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Examining the TF-IDF matrix\n\ntfidf_matrix.shape\ntfidf.get_feature_names_out()","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:53:27.391765Z","iopub.execute_input":"2023-09-30T11:53:27.392092Z","iopub.status.idle":"2023-09-30T11:53:27.405259Z","shell.execute_reply.started":"2023-09-30T11:53:27.392064Z","shell.execute_reply":"2023-09-30T11:53:27.404353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Cosine Similarity Calculation","metadata":{}},{"cell_type":"code","source":"# Calculating the similarity between TF-IDF vectors of documents using the cosine similarity metric\n\ncosine_sim = cosine_similarity(tfidf_matrix,tfidf_matrix)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:53:27.406425Z","iopub.execute_input":"2023-09-30T11:53:27.406799Z","iopub.status.idle":"2023-09-30T11:54:55.402766Z","shell.execute_reply.started":"2023-09-30T11:53:27.406774Z","shell.execute_reply":"2023-09-30T11:54:55.395738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Examining the cosine similarity metric\n\ncosine_sim.shape\ncosine_sim[1]","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:54:55.408807Z","iopub.execute_input":"2023-09-30T11:54:55.409672Z","iopub.status.idle":"2023-09-30T11:54:55.417814Z","shell.execute_reply.started":"2023-09-30T11:54:55.409623Z","shell.execute_reply":"2023-09-30T11:54:55.416767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Content-Based Recommender Function","metadata":{}},{"cell_type":"code","source":"def content_based_recommender(article_id, cosine_sim, dataframe):\n    indices = pd.Series(dataframe.index, index=dataframe['article_id'])\n    indices = indices[~indices.index.duplicated(keep='last')]\n    product_index = indices[article_id]\n    similarity_scores = pd.DataFrame(cosine_sim[product_index], columns=[\"score\"])\n    product_indices = similarity_scores.sort_values(\"score\", ascending=False).index[1:6]\n\n    return dataframe['article_id'].iloc[product_indices]","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:54:55.419626Z","iopub.execute_input":"2023-09-30T11:54:55.419951Z","iopub.status.idle":"2023-09-30T11:54:55.472186Z","shell.execute_reply.started":"2023-09-30T11:54:55.419924Z","shell.execute_reply":"2023-09-30T11:54:55.471044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The code snippet that generates recommendations for a random article using your content-based recommender\n\nmain_article_id = np.random.choice(df_ladies['article_id'])\n\nprint(main_article_id)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:54:55.474107Z","iopub.execute_input":"2023-09-30T11:54:55.475209Z","iopub.status.idle":"2023-09-30T11:54:55.49148Z","shell.execute_reply.started":"2023-09-30T11:54:55.475166Z","shell.execute_reply":"2023-09-30T11:54:55.490227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommended_articles = content_based_recommender(main_article_id, cosine_sim, df_ladies)\n\nprint(recommended_articles)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:54:55.49265Z","iopub.execute_input":"2023-09-30T11:54:55.493146Z","iopub.status.idle":"2023-09-30T11:54:55.520595Z","shell.execute_reply.started":"2023-09-30T11:54:55.493117Z","shell.execute_reply":"2023-09-30T11:54:55.519787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display recommended products and their images\n\nimage_path = \"../input/h-and-m-personalized-fashion-recommendations\"\n\ncols = 2\nrows = (len(recommended_articles) + cols - 1) // cols\n\n_df = df_ladies[df_ladies['article_id'].isin(recommended_articles)]\narticle_ids = _df.article_id.values[0:cols*rows]\n\nplt.figure(figsize=(2 + 3 * cols, 2 + 4 * rows))\nfor i in range(cols * rows):\n    plt.subplot(rows, cols, i + 1)\n    plt.axis('off')\n    \n    if i == 0:\n        article_id = (\"0\" + str(main_article_id))[-10:]\n        plt.title(f\"Main Article {article_id}\")\n    else:\n        article_id = (\"0\" + str(article_ids[i-1]))[-10:]\n        plt.title(f\"Recommended Article {article_id}\")\n    \n    image = Image.open(f\"{image_path}/images/{article_id[:3]}/{article_id}.jpg\")\n    plt.imshow(image)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-30T11:54:55.521649Z","iopub.execute_input":"2023-09-30T11:54:55.522114Z","iopub.status.idle":"2023-09-30T11:54:58.056414Z","shell.execute_reply.started":"2023-09-30T11:54:55.522086Z","shell.execute_reply":"2023-09-30T11:54:58.055438Z"},"trusted":true},"execution_count":null,"outputs":[]}]}