#############################################
#############################################
##
## H&M Previous Purchases + Most Popular
##
## All programming by
## John Akwei, ECMp ERMp Data Scientist
##
#############################################
#############################################

# This R language script recommends for customer
# future purchases the customer's most frequent
# purchases, in frequency order.

# If the customer has purchased less than 12 items
# from H&M, then the remaining suggestions are
# filled with the most popular items from the
# last two weeks.

# This script processes all 1,362,281 transactions.
# However, Kaggle does not provide the required
# processing power per script, to complete the
# processing in the requiste time span.

#############################################
#############################################
##
## Required Packages
##
#############################################
#############################################

suppressMessages({library(readr)
                  library(data.table)
                  library(datasets)
                  library(Matrix)
                  library(caret)
                  library(reshape2)})

#############################################
#############################################
##
## Data Import
##
#############################################
#############################################

transactions_train <-
fread("../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv")

#############################################
#############################################
##
## Format Data for Recommendation Processing
##
#############################################
#############################################

# Subset transactions_train for last two weeks
last_two_weeks <- transactions_train[transactions_train$t_dat > "2020-09-07" &
                                     transactions_train$t_dat < "2020-09-23",]

# Get the top 12 article ids in the last_two_weeks dataset
top_twelve_items <- {tmp <- table(last_two_weeks$article_id);
                     names(tmp)[order(tmp, decreasing = TRUE)][1:12]}

# Counter that loops through all transaction_train
# iter <- length(unique(transactions_train$customer_id))
iter <- 1362281

# Create dataframes for customer recommendation processing
new_submission <- data.frame(customer_id=NA, predictions=NA)
submission_df <- data.frame(customer_id=NULL, predictions=NULL)

# intialize counter for customer recommendation processing
i <- 14001

#############################################
#############################################
##
## Recommendation Processor
##
#############################################
#############################################

# Process 1,362,281 customers in sets of 10,000:

for(i in 14001:15000) {
    # Subset individual customer_id number
    next_customer <- transactions_train[as.numeric(as.factor(transactions_train$customer_id)) == i,]
    
    # Create dataset for individual recommendation processing
    transactions_recommendation <- data.frame(customer_number = i,
    article_id = as.numeric(next_customer$article_id),
    customer_id = as.factor(next_customer$customer_id))
    
    # Dataframe of customer ids/numbers and article ids
    user_item_data <- data.frame(customer_number = transactions_recommendation[,1],
                                 article_id = paste0("0", transactions_recommendation[,2]))
    
    # Create table
    user_item_table <- table(user_item_data)
    
    # Convert to dataframe
    user_item_table_df <- data.frame(user_item_table)
    
    # Order by Frequency
    user_item_table_df <- user_item_table_df[order(-user_item_table_df$Freq),]
    
    # If less than 12 customer item purchases, then
    # replace NAs in user_item_table_df$article_id with top 12 items
    for(i in 1:12) {
    if(nrow(user_item_table_df) < 12) {
        append_row <- data.frame(customer_number = user_item_table_df$customer_number,
                                 article_id = paste0("0",top_twelve_items[i]), Freq="1")
        user_item_table_df <- rbind(user_item_table_df, append_row[i,])
    }
    user_item_table_df
    }
    
    # Remove rownames
    rownames(user_item_table_df) <- NULL
    
    # Fill new submission row with next_customer data
    new_submission$customer_id <- next_customer$customer_id[1]
    
    # Create predictions array from frequency ordered table
    new_submission$predictions <- paste(user_item_table_df$article_id[1],
                                           user_item_table_df$article_id[2],
                                           user_item_table_df$article_id[3],
                                           user_item_table_df$article_id[4],
                                           user_item_table_df$article_id[5],
                                           user_item_table_df$article_id[6],
                                           user_item_table_df$article_id[7],
                                           user_item_table_df$article_id[8],
                                           user_item_table_df$article_id[9],
                                           user_item_table_df$article_id[10],
                                           user_item_table_df$article_id[11],
                                           user_item_table_df$article_id[12])
    
    # Row bind new customer with previous iterations
    submission_df <- rbind(submission_df, new_submission)
    
    # Output completed dataframe
    submission_df
}

# Format submission
write.csv(submission_df, "submission_14K_15K.csv", row.names=F)