{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Predicting tags from images \n\nIn this notebook, I use product features extracted by feeding products images into a pre-trained VGG16 CNN from Keras Applications to train a feed forward net to predict every label in the articles.csv file.\n<br>\n\nThe model is built using sklearn wrapper for keras and saved in h5 format in output folder.\n\nAll results and trained models are saved in pickle files as dictionaries with keys lableing the target variable and the trained model.\n<br>\n\nEmbeddings can be found here: https://www.kaggle.com/datasets/mohammedobeidat/hm-articlecustomer-embeddings-from-images\n<br>\nand generated in the notebook: https://www.kaggle.com/code/mohammedobeidat/h-m-customer-article-embeddings-from-images\n\nYou can set the target variable to choose which feature to predict.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pickle\nimport warnings\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\n\nfrom sklearn.linear_model import LogisticRegression as LR\nfrom sklearn.tree import DecisionTreeClassifier as DT\nfrom sklearn.ensemble import RandomForestClassifier as RF\nfrom sklearn.svm import SVC\n\nfrom sklearn.preprocessing import MinMaxScaler, LabelEncoder\nfrom sklearn.neighbors import KNeighborsClassifier as KNN\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import *\n\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Activation\nfrom keras.wrappers.scikit_learn import KerasClassifier\nfrom sklearn.model_selection import GridSearchCV\n\nwarnings.filterwarnings('ignore')\nplt.rcParams['figure.figsize'] = (8, 6)\nplt.rcParams['axes.grid'] = False\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T13:42:52.610039Z","iopub.execute_input":"2022-07-23T13:42:52.610513Z","iopub.status.idle":"2022-07-23T13:42:56.274802Z","shell.execute_reply.started":"2022-07-23T13:42:52.610419Z","shell.execute_reply":"2022-07-23T13:42:56.273546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = open('../input/hm-articlecustomer-embeddings-from-images/article_embeddings_from_image.pickle', 'rb')\nembeds = pickle.load(f)\nembeds = pd.DataFrame(embeds).reset_index()\nembeds.columns = ['article_id', 'embeds']\nembeds.embeds = embeds.embeds.map(np.array)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:42:56.276976Z","iopub.execute_input":"2022-07-23T13:42:56.277657Z","iopub.status.idle":"2022-07-23T13:43:00.027097Z","shell.execute_reply.started":"2022-07-23T13:42:56.277619Z","shell.execute_reply":"2022-07-23T13:43:00.026000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeds.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:43:00.029032Z","iopub.execute_input":"2022-07-23T13:43:00.029950Z","iopub.status.idle":"2022-07-23T13:43:00.054253Z","shell.execute_reply.started":"2022-07-23T13:43:00.029901Z","shell.execute_reply":"2022-07-23T13:43:00.053309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv', dtype={'article_id':str})\ndf = df.select_dtypes(object)\ndf = df.drop(['index_code', 'detail_desc', 'prod_name'], axis=1)\n\n# drop_mask = df.product_type_name.value_counts() > 5\n# to_keep = drop_mask[drop_mask == True].index\n# df = df[df.product_type_name.isin(to_keep)]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:43:00.056209Z","iopub.execute_input":"2022-07-23T13:43:00.057123Z","iopub.status.idle":"2022-07-23T13:43:00.898903Z","shell.execute_reply.started":"2022-07-23T13:43:00.057078Z","shell.execute_reply":"2022-07-23T13:43:00.897775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.merge(embeds, on='article_id')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:43:00.900207Z","iopub.execute_input":"2022-07-23T13:43:00.900555Z","iopub.status.idle":"2022-07-23T13:43:01.036173Z","shell.execute_reply.started":"2022-07-23T13:43:00.900508Z","shell.execute_reply":"2022-07-23T13:43:01.034768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:43:07.143026Z","iopub.execute_input":"2022-07-23T13:43:07.143812Z","iopub.status.idle":"2022-07-23T13:43:07.171598Z","shell.execute_reply.started":"2022-07-23T13:43:07.143775Z","shell.execute_reply":"2022-07-23T13:43:07.170076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def KLayerFeedForward():\n    clf = Sequential()\n    clf.add(Dense(1024, activation='relu', input_dim=1000))\n    clf.add(Dropout(0.1))\n    clf.add(Dense(2048, activation='relu'))\n    clf.add(Dropout(0.1))\n    clf.add(Dense(1024, activation='relu'))\n    clf.add(Dropout(0.1))\n    clf.add(Dense(512, activation='relu'))\n    clf.add(Dropout(0.1))\n    clf.add(Dense(128, activation='softmax'))\n    \n    \n    clf.compile(loss='BinaryCrossentropy', optimizer='sgd',\n                metrics=[\"accuracy\", tf.keras.metrics.Recall(), tf.keras.metrics.Precision()])\n    return clf\n\ncsv_logger = tf.keras.callbacks.CSVLogger('training.log')\n\nmodel = KerasClassifier(KLayerFeedForward, epochs=20, batch_size=64, verbose=1, callbacks=[csv_logger])\n\nmetrics = [accuracy_score, precision_score, recall_score]\nmetrics_names = ['accuracy_score', 'precision_score', 'recall_score']\n\nx = np.array(df['embeds'].to_list())\n\nsplit = int(0.8*len(x))\n\nx_train, x_test = x[:split], x[split:]\n\n\ntarget = 'product_type_name'\nmodel_name = 'Dense Network'\n\ny = np.array(df[target].to_list())\ny_train, y_test = y[:split], y[split:]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:43:57.064701Z","iopub.execute_input":"2022-07-23T13:43:57.065141Z","iopub.status.idle":"2022-07-23T13:43:57.399455Z","shell.execute_reply.started":"2022-07-23T13:43:57.065108Z","shell.execute_reply":"2022-07-23T13:43:57.398070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import np_utils\n\nencoder = LabelEncoder()\nencoder.fit(y)\n\nencoded_y_train = encoder.transform(y_train)\nencoded_y_test = encoder.transform(y_test)\n\n# convert integers to dummy variables (i.e. one hot encoded)\ndummy_y = np_utils.to_categorical(encoded_y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:43:57.745011Z","iopub.execute_input":"2022-07-23T13:43:57.745421Z","iopub.status.idle":"2022-07-23T13:43:57.887259Z","shell.execute_reply.started":"2022-07-23T13:43:57.745391Z","shell.execute_reply":"2022-07-23T13:43:57.886043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(x_train, dummy_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:43:58.305062Z","iopub.execute_input":"2022-07-23T13:43:58.306043Z","iopub.status.idle":"2022-07-23T13:44:04.187631Z","shell.execute_reply.started":"2022-07-23T13:43:58.305981Z","shell.execute_reply":"2022-07-23T13:44:04.185096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.DataFrame({'label': y_train}, index=df.article_id[:split])\ntest_df = pd.DataFrame({'label': y_test}, index=df.article_id[split:])\n\ntest_df['prediction'] = encoder.inverse_transform(model.predict(x_test))\ntrain_df['prediction'] = encoder.inverse_transform(model.predict(x_train))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_acc = accuracy_score(train_df['prediction'].values.reshape(-1, 1), y_train.reshape(-1, 1))\ntrain_precision = precision_score(train_df['prediction'].values.reshape(-1, 1), y_train.reshape(-1, 1), average='weighted')\ntrain_recall = recall_score(train_df['prediction'].values.reshape(-1, 1), y_train.reshape(-1, 1), average='weighted')\n\ntest_acc = accuracy_score(test_df['prediction'].values.reshape(-1, 1), y_test.reshape(-1, 1))\ntest_precision = precision_score(test_df['prediction'].values.reshape(-1, 1), y_test.reshape(-1, 1), average='weighted')\ntest_recall = recall_score(test_df['prediction'].values.reshape(-1, 1), y_test.reshape(-1, 1), average='weighted')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = pd.DataFrame({'accuracy':[train_acc, test_acc],\n                       'precision':[train_precision, test_precision],\n                       'recall':[train_recall, test_recall]}, index=['train', 'test']) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_y = pd.DataFrame({'label': y_test}, index=df.article_id[split:])[:1000]\nsample_x = x_test[:1000]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_y['prediction'] = encoder.inverse_transform(model.predict(sample_x))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_y.head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_caption():\n    \n    fig, axs = plt.subplots(1, 5, figsize=(20,10))\n    axs = axs.flatten()\n    \n    for ax in axs:\n        \n        sample = sample_y.sample(1)\n        img_path = f'../input/h-and-m-personalized-fashion-recommendations/images/{sample.index[0][:3]}/{sample.index[0]}.jpg'\n        img = plt.imread(img_path)\n        ax.imshow(img)\n        ax.axis('off')\n        txt = f'True Label: {sample.label.values[0]} \\n Random Forest: {sample.prediction.values[0]}'\n        ax.title.set_text(txt)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"generate_caption()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"generate_caption()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom skimage.transform import resize\nfrom PIL import Image\n!mkdir images\n\ndef plot_items(items):\n    path = \"../input/h-and-m-personalized-fashion-recommendations/images\"\n\n    k = len(items)\n    for item, i in zip(items, range(1, k+1)):\n        sub = item[:3]\n        image = path + \"/\"+ sub + \"/\"+ item +\".jpg\"\n        image = Image.open(image)\n       \n        basewidth = 360\n        wpercent = (basewidth / float(image.size[0]))\n        hsize = int((float(image.size[1]) * float(wpercent)))\n        image = image.resize((basewidth, hsize), Image.ANTIALIAS)\n        image.save(\"./images/{}.jpeg\".format(item[1:]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(sample_y.index.values)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.model.save('model.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_y.to_csv('product_tagging.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}