{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 목표 : \n1. article의 경우 10차원으로 표현이 가능하다(articles.csv파일에 product_type: 253 (product_type: vest top)처럼 제품군, 색에 대한 특성 벡터)\n따라서 ALS와 달리 customer만 벡터화 하고 상품은 위 특성을 이용하여 고정된 벡터로 표현해볼 수 있다.\n2. 가변 customer vector와 정적 article vector를 행렬 곱셈을 하고 실제 구매 데이터 테이블과 비교한다.\n* customer vector는 ALS의 경우 만들 수 있지만 지금처럼 article vector를 고정한 상태일 경우 만드는 상황이 아니다. DCGAN처럼 노이즈를 입력을 받는다.\n* 문제 손실함수 설정이 어려움(label이 customer에 대한 label이 아니고 article vector가 곱해진 것에 대한 라벨이라 반대방향으로 가중치 수정이 이루어질 수 있음)","metadata":{}},{"cell_type":"markdown","source":"# **DOWNLOAD DATA**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\n\nfname_tran ='../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv'\nfname_cus ='../input/h-and-m-personalized-fashion-recommendations/customers.csv'\nfname_article ='../input/h-and-m-personalized-fashion-recommendations/articles.csv'","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:54:18.241109Z","iopub.execute_input":"2022-02-16T10:54:18.241599Z","iopub.status.idle":"2022-02-16T10:54:18.269119Z","shell.execute_reply.started":"2022-02-16T10:54:18.241497Z","shell.execute_reply":"2022-02-16T10:54:18.268376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def loadData(filePAth):\n    return pd.read_csv(filePAth, sep=',')\n\ndata_cus = loadData(fname_cus)\ndata_article = loadData(fname_article)\ndata = loadData(fname_tran)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:54:18.270411Z","iopub.execute_input":"2022-02-16T10:54:18.270968Z","iopub.status.idle":"2022-02-16T10:55:31.388639Z","shell.execute_reply.started":"2022-02-16T10:54:18.270938Z","shell.execute_reply":"2022-02-16T10:55:31.387714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"using_cols = ['article_id', 'product_type_no', 'colour_group_code', 'index_group_no']\ndata_article_code = data_article[using_cols]","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:31.402537Z","iopub.execute_input":"2022-02-16T10:55:31.402839Z","iopub.status.idle":"2022-02-16T10:55:31.415360Z","shell.execute_reply.started":"2022-02-16T10:55:31.402804Z","shell.execute_reply":"2022-02-16T10:55:31.414468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **PREPROCESS DATA**","metadata":{}},{"cell_type":"code","source":"# 고유한 이름과 코드를 인덱스화합니다.\ndef initial_embedding(DataFrame, id_to_idx, targetColumn):\n\n    temp_data = DataFrame[targetColumn].map(id_to_idx.get).dropna()\n\n    if len(temp_data) == len(DataFrame):  \n        print('no-null')\n        DataFrame[targetColumn] = temp_data   \n    else:\n        print('detect null')","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:31.417424Z","iopub.execute_input":"2022-02-16T10:55:31.418373Z","iopub.status.idle":"2022-02-16T10:55:31.427132Z","shell.execute_reply.started":"2022-02-16T10:55:31.418333Z","shell.execute_reply":"2022-02-16T10:55:31.426217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 총 131개의 다른 값을 갖습니다. 이 벡터에 유독 크게 변동이 될 수 있으므로 벡터를 정규화하였습니다.\n# 특히 군집화를 통해 비슷한 article 그룹을 만들때 (0,0,0....)으로부터의 거리를 통해 군집화를 시도해보고자 정규화가 필요하다 생각했습니다.\n\ndata_article_code_unique = data_article_code.sort_values(by=['product_type_no'])['product_type_no'].unique()\ndata_article_code_to_idx = {v:k/131.0 for k,v in enumerate(data_article_code_unique)}\n\ninitial_embedding(data_article_code, data_article_code_to_idx, 'product_type_no')","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:31.428288Z","iopub.execute_input":"2022-02-16T10:55:31.428516Z","iopub.status.idle":"2022-02-16T10:55:31.510685Z","shell.execute_reply.started":"2022-02-16T10:55:31.428490Z","shell.execute_reply":"2022-02-16T10:55:31.509786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colour_group_code_unique = data_article_code.sort_values(by=['colour_group_code'])['colour_group_code'].unique()\ncolour_group_code_unique_to_idx = {v:k/49 for k,v in enumerate(colour_group_code_unique)}\n\ninitial_embedding(data_article_code, colour_group_code_unique_to_idx, 'colour_group_code')","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:31.512354Z","iopub.execute_input":"2022-02-16T10:55:31.512681Z","iopub.status.idle":"2022-02-16T10:55:31.583613Z","shell.execute_reply.started":"2022-02-16T10:55:31.512638Z","shell.execute_reply":"2022-02-16T10:55:31.582231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index_group_no_unique = data_article_code.sort_values(by=['index_group_no'])['index_group_no'].unique()\nindex_group_no_unique_to_idx = {v:k/26.0 for k,v in enumerate(index_group_no_unique)}\n\ninitial_embedding(data_article_code, index_group_no_unique_to_idx, 'index_group_no')","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:31.584797Z","iopub.execute_input":"2022-02-16T10:55:31.585002Z","iopub.status.idle":"2022-02-16T10:55:31.652853Z","shell.execute_reply.started":"2022-02-16T10:55:31.584975Z","shell.execute_reply":"2022-02-16T10:55:31.651993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articleCSV_unique = data_article_code.sort_values(by=['article_id'])['article_id'].unique()\narticle_code_to_idx = {v:k for k,v in enumerate(articleCSV_unique)}\n\ninitial_embedding(data_article_code, article_code_to_idx, 'article_id')","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:31.653873Z","iopub.execute_input":"2022-02-16T10:55:31.654109Z","iopub.status.idle":"2022-02-16T10:55:31.802557Z","shell.execute_reply.started":"2022-02-16T10:55:31.654079Z","shell.execute_reply":"2022-02-16T10:55:31.801751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 데이터프레임 열을 벡터로 바꿈\ndata_article_code['vector'] = [[data_article_code['product_type_no'][i],\\\n                                data_article_code['colour_group_code'][i],\\\n                               data_article_code['index_group_no'][i]] \\\n                               for i in range(len(data_article_code))]","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:31.803777Z","iopub.execute_input":"2022-02-16T10:55:31.804004Z","iopub.status.idle":"2022-02-16T10:55:34.399360Z","shell.execute_reply.started":"2022-02-16T10:55:31.803977Z","shell.execute_reply":"2022-02-16T10:55:34.398497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_code_to_vector = {data_article_code['article_id'][i]:data_article_code['vector'][i] for i in range(len(data_article_code)) }","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:34.402233Z","iopub.execute_input":"2022-02-16T10:55:34.402469Z","iopub.status.idle":"2022-02-16T10:55:36.022382Z","shell.execute_reply.started":"2022-02-16T10:55:34.402440Z","shell.execute_reply":"2022-02-16T10:55:36.021361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_article_code_data = data['article_id'].map(article_code_to_vector.get).dropna()\nprint(len(temp_article_code_data),len(data))\nif len(temp_article_code_data) == len(data):\n    print('no-null')\n    data['article_vector'] = temp_article_code_data\nelse:\n    print('detect null')\ndata","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:36.023517Z","iopub.execute_input":"2022-02-16T10:55:36.023752Z","iopub.status.idle":"2022-02-16T10:55:44.780432Z","shell.execute_reply.started":"2022-02-16T10:55:36.023723Z","shell.execute_reply":"2022-02-16T10:55:44.779662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_unique = data['customer_id'].unique()\narticle_unique = data['article_id'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:44.781748Z","iopub.execute_input":"2022-02-16T10:55:44.782075Z","iopub.status.idle":"2022-02-16T10:55:53.236535Z","shell.execute_reply.started":"2022-02-16T10:55:44.782027Z","shell.execute_reply":"2022-02-16T10:55:53.235755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_to_idx = {v:k for k,v in enumerate(user_unique)}\narticle_to_idx = {v:k for k,v in enumerate(article_unique)}","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:53.237474Z","iopub.execute_input":"2022-02-16T10:55:53.238156Z","iopub.status.idle":"2022-02-16T10:55:53.945834Z","shell.execute_reply.started":"2022-02-16T10:55:53.238123Z","shell.execute_reply":"2022-02-16T10:55:53.945193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initial_embedding(data, user_to_idx, 'customer_id')\n\ndata","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:55:53.968996Z","iopub.execute_input":"2022-02-16T10:55:53.969703Z","iopub.status.idle":"2022-02-16T10:56:16.239521Z","shell.execute_reply.started":"2022-02-16T10:55:53.969664Z","shell.execute_reply":"2022-02-16T10:56:16.238679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MODEL","metadata":{}},{"cell_type":"markdown","source":"# 목표:\n1. customer를 임베딩하는 레이어를 포함한 모델을 만든다.\n-> 행렬 곱을 하려면 article의 벡터 차원에 맞춰야하므로 출력을 1~10사이의 article차원에 맞춘다.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport time\nfrom scipy.sparse import csr_matrix","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:16.240805Z","iopub.execute_input":"2022-02-16T10:56:16.241182Z","iopub.status.idle":"2022-02-16T10:56:21.373743Z","shell.execute_reply.started":"2022-02-16T10:56:16.241152Z","shell.execute_reply":"2022-02-16T10:56:21.372828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"USER_COUNT = len(user_unique)\nARTICLE_COUNT = len(article_code_to_idx)\nDATA_NUM = data.shape[0]\n\noptimizer = tf.keras.optimizers.Adam(1e-4)\ncross_entropy = tf.keras.losses.BinaryCrossentropy(from_logits=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:21.375363Z","iopub.execute_input":"2022-02-16T10:56:21.375660Z","iopub.status.idle":"2022-02-16T10:56:22.543574Z","shell.execute_reply.started":"2022-02-16T10:56:21.375622Z","shell.execute_reply":"2022-02-16T10:56:22.542423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_customer = tf.keras.Sequential([\n    tf.keras.layers.Embedding(ARTICLE_COUNT, 64),\n    tf.keras.layers.Dense(128, activation='relu'),\n    tf.keras.layers.Dense(3)\n])","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:22.547784Z","iopub.execute_input":"2022-02-16T10:56:22.548245Z","iopub.status.idle":"2022-02-16T10:56:22.740069Z","shell.execute_reply.started":"2022-02-16T10:56:22.548210Z","shell.execute_reply":"2022-02-16T10:56:22.737718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 정적 데이터인 customer를 모델이 학습을 통해 처리한다기 보단 GAN과 같이 의미없는 데이터를 의미있게 만드는 것이 이 문제의\n# 핵심이라고 가정했습니다. 다만 노이즈를 입력으로 받을때 음수처리에 대한 문제가 있고,\n# 임베딩한 데이터를 임베팅 레이어(모델의 레이어)에 주입하는 거와 차이가 없는것 같아 아래에서는 사용하지는 않았습니다.\nnoise = tf.random.normal([DATA_NUM,1])","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:22.741444Z","iopub.execute_input":"2022-02-16T10:56:22.741668Z","iopub.status.idle":"2022-02-16T10:56:23.192074Z","shell.execute_reply.started":"2022-02-16T10:56:22.741642Z","shell.execute_reply":"2022-02-16T10:56:23.191240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# articles.csv파일의 article정보를 (3, article총 개수 )형태의 행렬로 만듭니다.\n# customer를 (1,3)의 행렬로 만들어 둘을 곱하면 한 customer가 article들에 대한 구매 가능성이 되고 이를 실제 구매 데이터에 비교하려고\n# 시도했습니다.\n\ny_article_input = [np.array(v) for k,v in enumerate(data_article_code['vector'])]\ny_article_input = np.array(y_article_input)\n\ndef transpose_matrix(matrix):\n    return matrix.T\ny_article_input = transpose_matrix(y_article_input)\n\ny_article_input.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:23.193215Z","iopub.execute_input":"2022-02-16T10:56:23.193610Z","iopub.status.idle":"2022-02-16T10:56:23.466107Z","shell.execute_reply.started":"2022-02-16T10:56:23.193569Z","shell.execute_reply":"2022-02-16T10:56:23.465266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Matrix = tf.matmul(model_customer(noise[5], training=True), y_article_input)\nMatrix","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:23.467393Z","iopub.execute_input":"2022-02-16T10:56:23.468547Z","iopub.status.idle":"2022-02-16T10:56:23.549339Z","shell.execute_reply.started":"2022-02-16T10:56:23.468452Z","shell.execute_reply":"2022-02-16T10:56:23.548441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoding = article_code_to_idx[data['article_id'][0]]\nencoding = tf.one_hot(encoding, ARTICLE_COUNT)\nencoding.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:23.550633Z","iopub.execute_input":"2022-02-16T10:56:23.551730Z","iopub.status.idle":"2022-02-16T10:56:23.559798Z","shell.execute_reply.started":"2022-02-16T10:56:23.551674Z","shell.execute_reply":"2022-02-16T10:56:23.559045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compare_loss( max_value_index, real_value_index ):\n    return cross_entropy(real_value_index,max_value_index)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:23.561139Z","iopub.execute_input":"2022-02-16T10:56:23.561458Z","iopub.status.idle":"2022-02-16T10:56:23.567877Z","shell.execute_reply.started":"2022-02-16T10:56:23.561418Z","shell.execute_reply":"2022-02-16T10:56:23.567047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 우려되는 점은 articles 행렬을 곱함으로서 기울기의 부호자체가 변하지 않을까 하는 것입니다. \n# 실제로 약간의 테스트성으로 학습을 해본 결과 음으로 증폭되버리고 있다는 것입니다.\n\n@tf.function\ndef train_step(customer_id, encoding):\n    \n    with tf.GradientTape() as tape:\n        x_user = model_customer(customer_id, training=True)\n        Matrix = tf.matmul(x_user, y_article_input)\n        loss = compare_loss(Matrix, encoding)\n\n    gradients = tape.gradient(loss, model_customer.trainable_variables)\n    optimizer.apply_gradients(zip(gradients, model_customer.trainable_variables))","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:23.569302Z","iopub.execute_input":"2022-02-16T10:56:23.569851Z","iopub.status.idle":"2022-02-16T10:56:23.578618Z","shell.execute_reply.started":"2022-02-16T10:56:23.569822Z","shell.execute_reply":"2022-02-16T10:56:23.577537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_customer.summary()","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:23.579824Z","iopub.execute_input":"2022-02-16T10:56:23.580242Z","iopub.status.idle":"2022-02-16T10:56:23.593122Z","shell.execute_reply.started":"2022-02-16T10:56:23.580184Z","shell.execute_reply":"2022-02-16T10:56:23.592180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_customer.compile(loss=tf.keras.losses.CategoricalCrossentropy(),\n#               optimizer=tf.keras.optimizers.Adagrad(0.5),\n#               metrics=['accuracy'],\n#              )","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:23.595103Z","iopub.execute_input":"2022-02-16T10:56:23.595689Z","iopub.status.idle":"2022-02-16T10:56:23.602064Z","shell.execute_reply.started":"2022-02-16T10:56:23.595641Z","shell.execute_reply":"2022-02-16T10:56:23.601211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers = data['customer_id']\ncustomers = np.array(customers)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:56:23.603307Z","iopub.execute_input":"2022-02-16T10:56:23.603522Z","iopub.status.idle":"2022-02-16T10:56:23.744979Z","shell.execute_reply.started":"2022-02-16T10:56:23.603488Z","shell.execute_reply":"2022-02-16T10:56:23.744091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 모델에 대한 테스트를 위해 article을 3차원 표현으로 줄였고, 학습하는 데이터도 줄였습니다. 1000명의 customer에 대해서만 10번 반복합니다.\n\nlimit = 1000\ndef train(data, epochs):\n    for epoch in range(epochs):\n        num = 0\n        time_count_1 = 0\n        starts = time.time()\n\n        for i in range(limit):\n            start = time.time()\n            encoding = [article_code_to_idx[data['article_id'][i]]]\n            encoding = tf.one_hot(encoding, ARTICLE_COUNT)\n            train_step(customers[tf.newaxis, i],encoding)\n \n            end = time.time()\n            time_count_1 = (end - start)\n            if num%100 == 0 :\n                print('.' , end = ' ')\n            if num%300 == 0:\n                time_left = (((limit-num) / 1) * time_count_1 / 60)\n                print(f\"{time_count_1:.5f} sec / TIME_LEFT(min): \",time_left)\n                time_count_1 = 0\n            num = num +1\n        # print (' 에포크 {} 에서 걸린 시간은 {} 초 입니다'.format(epoch +1, time.time()-start))\n        print ('Time for epoch {} is {} sec'.format(epoch + 1, time.time()-starts))","metadata":{"execution":{"iopub.status.busy":"2022-02-16T11:06:44.401099Z","iopub.execute_input":"2022-02-16T11:06:44.401426Z","iopub.status.idle":"2022-02-16T11:06:44.413886Z","shell.execute_reply.started":"2022-02-16T11:06:44.401390Z","shell.execute_reply":"2022-02-16T11:06:44.412906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nhistory = train(data, 10)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T11:06:46.870181Z","iopub.execute_input":"2022-02-16T11:06:46.870437Z","iopub.status.idle":"2022-02-16T11:14:06.832405Z","shell.execute_reply.started":"2022-02-16T11:06:46.870409Z","shell.execute_reply":"2022-02-16T11:14:06.831716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CHECK","metadata":{}},{"cell_type":"code","source":"def predict(num):\n    tem = tf.matmul(model_customer(customers[tf.newaxis, num], training=False), y_article_input)[0]\n    prediction_list = tf.math.argmax(tem)\n    return prediction_list,tem","metadata":{"execution":{"iopub.status.busy":"2022-02-16T11:14:19.298110Z","iopub.execute_input":"2022-02-16T11:14:19.298434Z","iopub.status.idle":"2022-02-16T11:14:19.303954Z","shell.execute_reply.started":"2022-02-16T11:14:19.298401Z","shell.execute_reply":"2022-02-16T11:14:19.303159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#첫번쨰 customer가 산 품목\nlist = data[data['customer_id']==0]['article_id'].to_numpy()\nlist","metadata":{"execution":{"iopub.status.busy":"2022-02-16T11:16:00.518337Z","iopub.execute_input":"2022-02-16T11:16:00.518821Z","iopub.status.idle":"2022-02-16T11:16:00.552904Z","shell.execute_reply.started":"2022-02-16T11:16:00.518770Z","shell.execute_reply":"2022-02-16T11:16:00.552308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction ,predicted_vector = predict(0)\n#모든 출력이 음수..\nprint(prediction, predicted_vector)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T11:16:04.826365Z","iopub.execute_input":"2022-02-16T11:16:04.826914Z","iopub.status.idle":"2022-02-16T11:16:04.836797Z","shell.execute_reply.started":"2022-02-16T11:16:04.826875Z","shell.execute_reply":"2022-02-16T11:16:04.836170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 38604로 모둔 입력에 대해 수렴하는 문제가 발생합니다.\npredicted_vector[38604]","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:14:25.969951Z","iopub.execute_input":"2022-02-16T10:14:25.970452Z","iopub.status.idle":"2022-02-16T10:14:25.976815Z","shell.execute_reply.started":"2022-02-16T10:14:25.9704Z","shell.execute_reply":"2022-02-16T10:14:25.97624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#아이디값을 다시 article_id로 변환\nidx_to_article_code = {v:k for k,v in article_code_to_idx.items()}\nidx_to_article_code[38604]","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:07:08.920586Z","iopub.execute_input":"2022-02-16T10:07:08.9212Z","iopub.status.idle":"2022-02-16T10:07:08.936629Z","shell.execute_reply.started":"2022-02-16T10:07:08.921161Z","shell.execute_reply":"2022-02-16T10:07:08.935965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**check most suitable item**","metadata":{}},{"cell_type":"code","source":"from IPython.display import Image","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k = [658223002]\n\nnum = 0\nImage(f'../input/h-and-m-personalized-fashion-recommendations/images/0{str(k[num])[:2]}/0{int(k[num])}.jpg' , width = 200)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:05:10.958053Z","iopub.execute_input":"2022-02-16T10:05:10.958348Z","iopub.status.idle":"2022-02-16T10:05:10.962694Z","shell.execute_reply.started":"2022-02-16T10:05:10.958311Z","shell.execute_reply":"2022-02-16T10:05:10.961919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k = list\n\nnum = 0 #check what the customer bought 0<= num <=9\nImage(f'../input/h-and-m-personalized-fashion-recommendations/images/0{str(k[num])[:2]}/0{int(k[num])}.jpg' , width = 200)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T10:05:14.912052Z","iopub.execute_input":"2022-02-16T10:05:14.912511Z","iopub.status.idle":"2022-02-16T10:05:14.926738Z","shell.execute_reply.started":"2022-02-16T10:05:14.912456Z","shell.execute_reply":"2022-02-16T10:05:14.925942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ANOTHER MODEL","metadata":{}},{"cell_type":"code","source":"# TATAL_NUM = int(len(data)/10)\n\n# user_unique = data['customer_id'].unique()\n# user_to_idx = {v:k for k,v in enumerate(user_unique)}\n# temp_user_data = data['customer_id'].map(user_to_idx.get).dropna()\n\n# if len(temp_user_data) == len(data):  \n#     print('no-null')\n#     data['customer_id'] = temp_user_data   \n# else:\n#     print('detect null')\n# data","metadata":{"execution":{"iopub.status.busy":"2022-02-16T07:14:56.142956Z","iopub.execute_input":"2022-02-16T07:14:56.143677Z","iopub.status.idle":"2022-02-16T07:15:21.830594Z","shell.execute_reply.started":"2022-02-16T07:14:56.143636Z","shell.execute_reply":"2022-02-16T07:15:21.829619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data = data[:TATAL_NUM]\n# ##\n# starts = time.time()\n# train_label = [data_article_id_to_idx[data['article_id'][i]] for i in range(TATAL_NUM)]\n# end = time.time()\n# ##\n# time_count_1 = (end - starts)\n# time_left = time_count_1\n# time_left = time_left\n# print(f\"{time_count_1:.5f} sec / TIME_LEFT(min): \",time_left)\n# from sklearn.model_selection import train_test_split\n# train_input, val_input, train_label, val_label = \\\n#     train_test_split(train_data['customer_id'].to_numpy(), train_label, shuffle=True, test_size = 0.1)\n# ##\n# def get_matrix_factorization_model():\n    \n#     item_input = tf.keras.layers.Input(shape=[1], name='Item')\n#     item_embedding_layer = tf.keras.layers.Embedding(\n#       1000,\n#       64,\n#       name='ItemEmbedding')\n    \n#     x = tf.keras.layers.Dense(128, activation='relu')(tf.cast(item_embedding_layer(item_input),tf.int64))\n#     print(tf.cast(item_embedding_layer(item_input),tf.int64))\n#     x1 = tf.keras.layers.Dense(3)(x)\n    \n#     pred = tf.keras.layers.Dot(\n#     (2,1), name='Dot')([x1, tf.expand_dims(y_article_input,0)])\n    \n#     pred = tf.cast(pred, tf.int64)\n#     pred = tf.math.argmax(pred)\n#     pred = tf.cast(pred, tf.int64)\n    \n#     model = tf.keras.Model(inputs=item_input, outputs=pred)\n\n#     return model\n\n# ##\n# model_a = tf.keras.Sequential([\n#     tf.keras.layers.Flatten(input_shape=(105542, )),\n#     tf.keras.layers.Dense(128, activation='relu'),\n#     tf.keras.layers.Dense(3)\n# ])\n# ##\n# model = get_matrix_factorization_model()","metadata":{"execution":{"iopub.status.busy":"2022-02-16T07:28:16.046021Z","iopub.execute_input":"2022-02-16T07:28:16.046737Z","iopub.status.idle":"2022-02-16T07:28:16.054574Z","shell.execute_reply.started":"2022-02-16T07:28:16.046695Z","shell.execute_reply":"2022-02-16T07:28:16.05369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.compile(loss=tf.keras.losses.SparseCategoricalCrossentropy(),\n#               optimizer=tf.keras.optimizers.Adagrad(0.5),\n#               metrics=['accuracy'],\n#              )\n# ##\n# model.summary()\n# ##\n# train_input = np.array(train_input)\n# train_label = np.array(train_label)\n# val_input = np.array(val_input)\n# val_label = np.array(val_label)\n# ##\n# history = model.fit(train_input, train_label, epochs=1,\n#                     validation_data=(val_input, val_label),\n#                     steps_per_epoch=100,\n#                     validation_steps=30)","metadata":{"execution":{"iopub.status.busy":"2022-02-16T07:28:45.13287Z","iopub.execute_input":"2022-02-16T07:28:45.133569Z","iopub.status.idle":"2022-02-16T07:28:45.144507Z","shell.execute_reply.started":"2022-02-16T07:28:45.133526Z","shell.execute_reply":"2022-02-16T07:28:45.143634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n\n# def plot_graphs(history, metric):\n#     plt.plot(history.history[metric])\n#     plt.plot(history.history['val_'+metric], '')\n#     plt.xlabel(\"Epochs\")\n#     plt.ylabel(metric)\n#     plt.legend([metric, 'val_'+metric])\n\n# plt.figure(figsize=(8, 4))\n# plt.subplot(1, 2, 1)\n# plot_graphs(history, 'accuracy')\n# plt.ylim(None, 1)\n# plt.subplot(1, 2, 2)\n# plot_graphs(history, 'loss')\n# plt.ylim(0, None)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **SUBMISSION**","metadata":{}}]}