# Two questions:
# 1. Can you improve on this benchmark?
# 2. Can you beat the score obtained by this example kernel?

import pandas as pd
import numpy as np
from sklearn.metrics import log_loss
from scipy.optimize import minimize
from nltk.corpus import stopwords
from nltk.tokenize import RegexpTokenizer
from nltk.stem.snowball import SnowballStemmer

stops = set(stopwords.words("english"))
tokenizer = RegexpTokenizer('\w+')
stemmer = SnowballStemmer("english", ignore_stopwords=True)

sentence_to_words = lambda s: set(map(stemmer.stem, tokenizer.tokenize(s))) - stops

def word_match_share(row):
    q1words = sentence_to_words(str(row['question1']))
    q2words = sentence_to_words(str(row['question2']))

    if len(q1words) == 0 or len(q2words) == 0:
        # The computer-generated chaff includes a few questions that are nothing but stopwords
        return 0
    R = (len(q1words & q2words))/(len(q1words | q2words))
    return R

#LOAD TRAINSET, TESTSET
train = pd.read_csv("../input/train.csv")
train[ 'R' ] = train.apply( word_match_share, axis=1, raw=True )
print( train.head() ) 

test = pd.read_csv("../input/test.csv", index_col=False )
test['R'] = test.apply( word_match_share, axis=1, raw=True )
print( test.head() )

#Mean target
GLOBAL_MEAN = np.mean( train['is_duplicate'] ) 
print( 'Mean is_duplicated', GLOBAL_MEAN )

#OPTIMIZE FUNCTIONS
def minimize_train_log_loss( W ):
    train["prediction"] = train["R"] * W[0] + W[1]
    score = log_loss( train['is_duplicate'], train['prediction'] )
    print(  score , W )
    return( score )

res = minimize(minimize_train_log_loss, [0.00,  0.00], method='Nelder-Mead', tol=1e-4, options={'maxiter': 400})
W = res.x
print( 'Best weights: ',W )


#APPLY TO TESTSET
test["is_duplicate"] = test["R"] * W[0] + W[1]
test[ ['test_id','is_duplicate'] ].to_csv("count_words_benchmark.csv", header=True, index=False)

