{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\n########################################################################################################################\n# 用神经网络中DNN构建文本分类模型,具体而言分为两个部分：\n# ·数据处理\n# ·DNN模型:输入层,隐层,sigmoid输出层\n########################################################################################################################\n# question统一调整为 70 个 word,不足的填充<pad>，长的则截断\n# vocabulary 大小为 <= 20万 , 且去除 词频大于2000的词\n# DNN含1个隐藏层:隐层神经元个数512\n# version3 : probability >= Threshold,为 1 ；probability < Threshold , 为 0\n# 本版本用130万条数据来训练,Threshold:0.5,相比version3,probability > Threshold,为 1 ；probability <= Threshold , 为 0\n# 本版本做了如下改进：\n# 原数据大约有130万条,但 insincere questions只有80442条,而sincere questions却有1219559条,insincere questions 占 sincere questions的比例只有 7%\n# 将insincere questions and sincere questions 的比例设置为1:5,考虑到了样本不均衡给模型学习带来的影响\n# 对 sincere questions随机采样时,是在全量sincere questions数据集上进行的\n# 将Accuracy,Precision,Recall,F1_Score在训练过程中的变化以图表的形式展现出来了\n# 增加迭代次数:EPOCHES=50--->EPOCHES=100\n# 增加了隐藏层,共三层隐藏层,HIDDEN_SIZE = [512,256,128];启用Dropout\n# 使用所有词","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# Data_stage_1.py\n# -*- coding: utf-8 -*-\n# @Time    : 2019/1/7 14:56\n# @Author  : Weiyang\n\n'''\n目标：读取train.csv和test.csv数据，去除标点符号等特殊符号，将question变为word列表，单词转为小写、\n输入：train.csv,test.csv\n输出1：questions = [[word1,word2,word3,..],...]\n输出2：labels = [1,0,1,...]\n输出3：test_questions = [[word1,word2,word3,..],...]\n输出4：test_qids = [001256,05891,...]\n输出5：empty_question_qids = [] # 如果去除特殊符号后,test.csv中question为空,则question类别为1,即是insincere question;其不再参与预测\n注意:若去除特殊符号后,question为空,则将question直接删除,其不参与训练；而如果test.csv中出现此情况,则question类别确定为1,也无需参与预测;\n'''\n\nimport pandas as pd\nimport re\nimport time\nimport numpy as np\n\nstart = time.time()\n# read train.csv\ntrain_data = pd.read_csv('../input/train.csv',encoding='utf-8-sig')\n\n# 去除标点符号等特殊符号，将question变为词列表,将单词转为小写\nquestions = [] # [[word1,word2,..],..] question文本且分成一个一个词\nlabels = [] # labels\nsymbols = ['\\''] # don't, I'm\nrandint = np.random.randint(0,len(train_data['question_text']),100) # 产生100个随机整数\ncount = 0\nfor question,label in zip(train_data['question_text'],train_data['target']):\n    if count in randint:\n        print('train.csv: ',question)\n    question = re.sub(\"[\\s+\\.\\!\\/_,\\\\:;><{}\\-$%^*()+\\\"\\[\\]]+|[+——！，。：；》《？?、~@#￥%……&*（）]+\",' ',question)\n    question = [str(word) for word in question.split() if len(word.strip()) !=0 and word not in symbols]\n    question = [word.rstrip('\\'').lstrip('\\'') for word in question] # 去除左右两侧的单引号\n    question = [word.rstrip('\\\\').lstrip('\\\\') for word in question]  # 去除左右两侧的\\\\\n    question = [word.rstrip('’').lstrip('’') for word in question]  # 去除左右两侧的’\n    question = [word.rstrip('‘').lstrip('‘') for word in question]  # 去除左右两侧的‘\n    question = [word.rstrip('”').lstrip('”') for word in question]  # 去除左右两侧的”\n    question = [word.rstrip('“').lstrip('“') for word in question]  # 去除左右两侧的“\n    question = [word.rstrip('`').lstrip('`') for word in question]  # 去除左右两侧的`\n    #question = [word.lower() for word in question]  # 将单词统一为小写\n    \n    # 将question首个单词变为小写，其余单词形式不变\n    temp = []\n    for i in range(len(question)):\n        if i == 0:\n            temp.append(question[i].lower())\n        else:\n            temp.append(question[i])\n    question = temp[:]\n    \n    # 判断question是否为空\n    if len(question) == 0:\n        continue\n    questions.append(question)\n    labels.append(label)\n    if count in randint:\n        print('train.csv: ',question)\n    count += 1\n\ndel train_data\n\n# read test.csv\ntest_data = pd.read_csv('../input/test.csv',encoding='utf-8-sig')\n\n# 去除标点符号等特殊符号，将question变为词列表,将单词转为小写\ntest_questions = [] # [[word1,word2,..],..] question文本且分成一个一个词\ntest_qids = [] #qids\nempty_question_qids = [] # 如果去除特殊符号后,question为空,则question类别为1,即是insincere question;其不再参与预测\nsymbols = ['\\''] # don't,I'm,...\nrandint = np.random.randint(0,len(test_data['question_text']),100) # 产生100个随机整数\ncount = 0\nfor question,qid in zip(test_data['question_text'],test_data['qid']):\n    #if count > 3000:\n        #break\n    if count in randint:\n        print('test.csv: ',question)\n    question = re.sub(\"[\\s+\\.\\!\\/_,\\\\:;><{}\\-$%^*()+\\\"\\[\\]]+|[+——！，。：；》《？?、~@#￥%……&*（）]+\",' ',question)\n    question = [str(word) for word in question.split() if len(word.strip()) !=0 and word not in symbols]\n    question = [word.rstrip('\\'').lstrip('\\'') for word in question] # 去除左右两侧的单引号\n    question = [word.rstrip('\\\\').lstrip('\\\\') for word in question]  # 去除左右两侧的\\\\\n    question = [word.rstrip('’').lstrip('’') for word in question]  # 去除左右两侧的’\n    question = [word.rstrip('‘').lstrip('‘') for word in question]  # 去除左右两侧的‘\n    question = [word.rstrip('”').lstrip('”') for word in question]  # 去除左右两侧的”\n    question = [word.rstrip('“').lstrip('“') for word in question]  # 去除左右两侧的“\n    question = [word.rstrip('`').lstrip('`') for word in question]  # 去除左右两侧的`\n    #question = [word.lower() for word in question]  # 将单词统一为小写\n    \n    # 将question首个单词变为小写，其余单词形式不变\n    temp = []\n    for i in range(len(question)):\n        if i == 0:\n            temp.append(question[i].lower())\n        else:\n            temp.append(question[i])\n    question = temp[:]\n    \n    # 判断question是否为空\n    if len(question) == 0:\n        empty_question_qids.append(qid)\n        continue\n    test_questions.append(question)\n    test_qids.append(qid)\n    if count in randint:\n        print('test.csv: ',question)\n    count += 1\n\ndel test_data\n\nprint('The total time of Data_stage_1.py program is : %d s' %(time.time()-start))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f04dd004d2fac26d6cbe7c6cb971b392e14cd606"},"cell_type":"code","source":"# Data_stage_2.py\n# -*- coding: utf-8 -*- \n# @Time    : 2019/1/13 10:23 \n# @Author  : Weiyang\n\n'''\n目标：数据探索+预处理\n输入：questions,labels,test_questions\n输出1：\n'''\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nfrom collections import Counter\nimport tqdm\nimport time\n\nstart = time.time()\n\n# ------------------------------------------------数据探索：基本的统计分析------------------------------------------------\n\n'将train.csv分成insincere question 和 sincere question 两部分数据'\n# 加载insincere question\n# insincere question: 1 , sincere question: 0\ninsincere_questions = []\nsincere_questions = [] \nfor i in range(len(questions)):\n    if labels[i] == 0:\n        sincere_questions.append(questions[i])\n    elif labels[i] == 1:\n        insincere_questions.append(questions[i])\n\n# 数据探索：描述性统计\n# insincere questions and sincere questions 类别的比例\nprint('The Proportion of Insincere and sincere questions : %.2f'%(float(len(insincere_questions)/len(sincere_questions))))\n# insincere questions 统计\nprint('-' * 30 + 'Insincere Questions' + '-' * 30)\nprint('Total number of insincere questions : {}'.format(len(insincere_questions)))\nprint('The average length of insincere questions: {}'.format(np.mean([len(question) for question in insincere_questions])))\nprint('The max length of insincere questions : {}'.format(np.max([len(question) for question in insincere_questions])))\nprint('The min length of insincere questions : {}'.format(np.min([len(question) for question in insincere_questions])))\n# 统计高频词\nc = Counter([word for question in insincere_questions for word in question]).most_common(100)\nprint('Most common words in insincere questions : \\n{}'.format(c))\n\n# sincere questions 统计\nprint('-' * 30 + 'Sincere Questions' + '-' * 30)\nprint('Total number fo sincere questions : {}'.format(len(sincere_questions)))\nprint('The average length of sincere questions: {}'.format(np.mean([len(question) for question in sincere_questions])))\nprint('The max length of sincere questions : {}'.format(np.max([len(question) for question in sincere_questions])))\nprint('The min length of sincere questions : {}'.format(np.min([len(question) for question in sincere_questions])))\n# 统计高频词\nc = Counter([word for question in sincere_questions for word in question]).most_common(100)\nprint('Most common words in sincere questions : \\n{}'.format(c))\n\n# --------------------------------------------------数据预处理----------------------------------------------------------\n# ·构造词典Vocabulary\n# ·构造映射表\n# ·转换单词为tokens\n\n# 句子最大长度\nSENTENCE_LIMIT_SIZE = 70\n\n# 构造词典\n# 我们要基于整个语料来构造我们的词典，由于文本中包含许多干扰词汇，例如仅出现过1次的这类单词。对于这类极其低频词汇，我们可以对其\n# 进行去除，一方面能加快模型执行效率，一方面也能减少特殊词带来的噪声。\n\n# 将 questions+test_questions 扩展成单层列表:[word1,word2,....]\ntotal_words = [word for question in (questions+test_questions) for word in question]\n# 统计词汇\nc = Counter(total_words)\n# 倒序查看词频\nprint('The word frequency of train.csv and test.csv is : ')\nprint()\nsorted(c.most_common(),key=lambda x : x[1],reverse=True)\n\n# 初始化两个token: pad 和 unk\nvocab = ['<pad>','<unk>']\nvocab_max_length = 200000 # 词包最多能存储的单词数量\n\n# 去除出现频次大于word_frequency的单词\nword_frequency = 1\ncount = 1\nfor w,f in c.most_common():\n    if count > vocab_max_length:\n        break\n    if f > word_frequency:\n        vocab.append(w)\n    count += 1\nprint('The size of vocabulary is : {}'.format(len(vocab)))\n\n#构造映射\n# 单词到编码的映射，例如：machine ---> 10256\nword_to_token = {word: token for token,word in enumerate(vocab)}\n# 编码到单词的映射，例如：10256---> machine\ntoken_to_word = {token : word for token ,word in enumerate(vocab)}\n\n# 转换文本：对文本进行编码\ndef convert_text_to_token(sentence,word_to_token_map=word_to_token,limit_size=SENTENCE_LIMIT_SIZE):\n    '''\n    根据单词--编码映射表 将单个句子转化为token\n    :param sentence: 句子,类型是word list,[word1,word2,...]\n    :param word_to_token_map: 单词到编码的映射\n    :param limit_size: 句子最大长度。超过该长度的句子进行截断，不足的句子进行pad补全\n    :return: 句子转换为token后的列表\n    '''\n    # 获取 unknow 单词和 pad的token\n    unk_id = word_to_token_map['<unk>']\n    pad_id = word_to_token_map['<pad>']\n\n    # 对句子进行token转换，对于未在词典中出现过的词用unk的token填充\n    tokens = [word_to_token_map.get(word,unk_id) for word in sentence]\n    # 对句子长度进行规整，短的补全pad，长的截断 trunc\n    # pad\n    if len(tokens) < limit_size:\n        tokens.extend([pad_id]*(limit_size - len(tokens)))\n    # Trunc\n    else:\n        tokens = tokens[:limit_size]\n\n    return tokens\n\n# sincere questions随机采样,采样数量等于 insincere questions\nprint('Sampling sincere questions...')\nsincere_insincere_proportion = 4 # The ratio of sincere/insincere \nindex = np.random.randint(0,len(sincere_questions),int(len(insincere_questions)*sincere_insincere_proportion))\nsincere_questions = np.array(sincere_questions)[index]\nsincere_questions = sincere_questions.tolist()\nquestions = sincere_questions[:] + insincere_questions[:]\nlabels = [0]*len(sincere_questions) + [1]*len(insincere_questions)\ndel insincere_questions,sincere_questions\n# 混洗数据\nprint('Shuffling questions....')\nshuffled_index = np.random.permutation(range(len(labels)))\nquestions = np.array(questions)[shuffled_index]\nquestions = questions.tolist()\nlabels = np.array(labels)[shuffled_index]\nlabels = labels.tolist()\n\n# 将 questions 转为 编码列表 : [[1,5,9,55,..],....]\nprint('Begining convert questions to tokens list...')\nquestions_tokens = []\nfor question in tqdm.tqdm(questions):\n    tokens = convert_text_to_token(question)\n    questions_tokens.append(tokens)\nprint('Del questions...')\ndel questions\n# 将 test_questions 转为 编码列表 : [[1,5,9,55,..],....]\nprint('Begining convert test questions to tokens list...')\ntest_questions_tokens = []\nfor question in tqdm.tqdm(test_questions):\n    tokens = convert_text_to_token(question)\n    test_questions_tokens.append(tokens)\nprint('Del test questions...')\ndel test_questions\n\n# 转换为numpy格式,方便处理\nquestions_tokens = np.array(questions_tokens)\ntest_questions_tokens = np.array(test_questions_tokens)\nlabels = np.array(labels).reshape(-1,1)\n\nprint('The shape of all question tokens in our corpus: ({},{})'.format(*questions_tokens.shape))\nprint('The shape of all targets in our corpus : ({},)'.format(*labels.shape))\nprint('The total time of Data_stage_2.py program is : %d s' %(time.time()-start))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eabdb5bf5f9d296322bf55b342ae6c3aa0130e39"},"cell_type":"code","source":"# Data_stage_3.py\n# -*- coding: utf-8 -*- \n# @Time    : 2019/1/13 12:32 \n# @Author  : Weiyang\n\n# ------------------------------------------------构造词向量-------------------------------------------------------------\n# 这里使用 GoogleNews-vectors-negative300.bin预训练好的词向量来做embedding:\n# ·如果当前词没有对应的词向量，则用随机数产生的向量替代\n# ·如果当前词为 <PAD> ,则用 0向量替代\n\nimport warnings\nwarnings.filterwarnings(action=\"ignore\")\nimport numpy as np\nfrom gensim.models.keyedvectors import KeyedVectors\nimport time\nimport sys\n\nstart = time.time()\n\n# loads 300x1 word vectors from file.\ndef load_bin_vec(fname):\n    words_vector = KeyedVectors.load_word2vec_format(fname, binary=True) # 通过model来取对应词的词向量\n    return words_vector\n# 预训练的词向量的路径\nvectors_file =  '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\nprint('Loading all_words ....')\nprint('Loading googleNew vector....')\nwords_vector = load_bin_vec(vectors_file)  # pre-trained vectors\ngoogle_words = set() # 存储 GoogleNews-vectors-negative300 中的 词\nword_to_vec = {} # GoogleNews-vectors-negative300: {word:vector,...}\nfor word in tqdm.tqdm(words_vector.wv.vocab.keys()):\n    google_words.add(word)\n    word_to_vec[word] = np.array(words_vector[word],dtype=np.float32)     \nprint('The number of words which have pretrained-vectors in vocab is : {}'.format(len(set(vocab)&set(google_words))))\nprint()\nprint('The number of words which do not have pretrained-vectors in vocab is : {}'.format(len(set(vocab))-len(set(vocab)&set(google_words))))\ndel google_words,words_vector,vectors_file\n\n\n# 构造词向量矩阵\nVOCAB_SIZE = len(vocab) # \nEMBEDDING_SIZE = 300\n\n# 初始化词向量矩阵(这里命名为 static是因为这个词向量矩阵用预训练好的填充，无需重新训练)\nstatic_embeddings = np.zeros([VOCAB_SIZE,EMBEDDING_SIZE])\nfor word,token in tqdm.tqdm(word_to_token.items()):\n    # 用google_vector词向量填充，如果没有对应的词向量，则用随机数填充\n    word_vector = word_to_vec.get(word,0.2 * np.random.random(EMBEDDING_SIZE) - 0.1)\n    static_embeddings[token,:] = word_vector\ndel word_to_vec\n    \n# 重置PAD为0向量\npad_id = word_to_token['<pad>']\nstatic_embeddings[pad_id,:] = np.zeros(EMBEDDING_SIZE)\nstatic_embeddings = static_embeddings.astype(np.float32)\n\n# --------------------------------------------分割训练集和测试集---------------------------------------------------------\n\ndef split_train_test(x,y,train_ratio=0.8,shuffle=True):\n    '''\n    分割train 和 test\n    :param x: 输入特征序列\n    :param y: 标签序列\n    :param train_ratio: 训练样本比例\n    :param shuffle: 是否shuffle\n    :return:\n    '''\n    assert x.shape[0] == y.shape[0] , print('error shape!')\n\n    if shuffle:\n        shuffled_index = np.random.permutation(range(x.shape[0]))\n        x = x[shuffled_index]\n        y = y[shuffled_index]\n    # 分离 train 和 test\n    train_size = int(x.shape[0] * train_ratio)\n    x_train = x[:train_size]\n    x_test = x[train_size:]\n    y_train = y[:train_size]\n    y_test = y[train_size:]\n\n    return x_train,x_test,y_train,y_test\n\n# 划分train 和 test\nx_train,x_test,y_train,y_test = split_train_test(questions_tokens,labels)\nprint('Del questions_token and labels ...')\ndel questions_tokens,labels\n\n# ----------------------------------------------批量获取数据-------------------------------------------------------------\n\ndef get_batch(x,y,batch_size=300,shuffle=True):\n    assert x.shape[0] == y.shape[0],print('error shape!')\n    # shuffle\n    if shuffle:\n        shuffled_index = np.random.permutation(range(x.shape[0]))\n        x = x[shuffled_index]\n        y = y[shuffled_index]\n    # 统计共有几个完整的batch\n    n_batches = int(x.shape[0] / batch_size)\n    for i in range(n_batches - 1):\n        x_batch = x[i * batch_size:(i+1)*batch_size]\n        y_batch = y[i * batch_size:(i+1)*batch_size]\n        yield x_batch,y_batch\n        \nprint('The total time of Data_stage_3.py program is : %d s' %(time.time()-start))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"31f8bfba8ffc2c867ae4bbd242db35e4ee602ba4"},"cell_type":"code","source":"# DNN.py\n# -*- coding: utf-8 -*- \n# @Time    : 2018/11/16 10:27 \n# @Author  : ZhangJiaLin\n\n########################################################################################################################\n# DNN 模型实现 Quora insincere questions classification\n# HIDDEN_SIZE : 512\n# layer_size : 4\n########################################################################################################################\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport tensorflow as tf\nfrom sklearn.metrics import precision_score\nfrom sklearn.metrics import recall_score\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import accuracy_score\nimport numpy as np\nimport time\n\n# 定义神经网络超参数\nHIDDEN_SIZE = [512,256,128]\nLEARNING_RAGE = 0.001\nEPOCHES = 100\nBATCH_SIZE = 256\nKEEP_PROB = 0.4 # Dropout层保留概率\nL2_lamda = 0.1 # L2正则化lamda\nthreshold = 0.36 # 阈值\n\nstart = time.time()\n\nwith tf.name_scope('DNN'):\n    # 输入及输出tensor\n    with tf.name_scope('placeholders'):\n        inputs = tf.placeholder(dtype=tf.int32,shape=(None,SENTENCE_LIMIT_SIZE),name='inputs')\n        targets = tf.placeholder(dtype=tf.float32,shape=(None,1),name='targets')\n\n    # embeddings\n    with tf.name_scope('embeddings'):\n        # 用 pre-trained 的词向量来作为embedding层\n        embedding_matrix = tf.Variable(initial_value=static_embeddings,trainable=False,name='embedding_matrix')\n        embed = tf.nn.embedding_lookup(embedding_matrix,inputs,name='embed')\n        # 将句子中每个词的词向量对应维度的值累加得到句子向量\n        sum_embed = tf.reduce_sum(embed,axis=1,name='sum_embed')\n\n    # model\n    with tf.name_scope('model'):        \n        # 隐层第一层权重\n        W1 = tf.Variable(tf.random_normal(shape=(EMBEDDING_SIZE, HIDDEN_SIZE[0]), stddev=0.1), name='W1')\n        b1 = tf.Variable(tf.zeros(shape=(HIDDEN_SIZE[0]), name='b1'))\n\n        # 隐层第二层权重\n        W2 = tf.Variable(tf.random_normal(shape=(HIDDEN_SIZE[0], HIDDEN_SIZE[1]), stddev=0.1), name='W2')\n        b2 = tf.Variable(tf.zeros(shape=(HIDDEN_SIZE[1]), name='b2'))\n\n        # 隐层第三层权重\n        W3 = tf.Variable(tf.random_normal(shape=(HIDDEN_SIZE[1], HIDDEN_SIZE[2]), stddev=0.1), name='W2')\n        b3 = tf.Variable(tf.zeros(shape=(HIDDEN_SIZE[2]), name='b3'))\n\n        # 输出层权重\n        W4 = tf.Variable(tf.random_normal(shape=(HIDDEN_SIZE[2], 1), stddev=0.1), name='W2')\n        b4 = tf.Variable(tf.zeros(shape=(1), name='b4'))\n\n        # 隐层第一层输出\n        z1 = tf.add(tf.matmul(sum_embed, W1), b1)\n        # Dropout层\n        z1 = tf.nn.dropout(z1,keep_prob=KEEP_PROB)\n        a1 = tf.nn.relu(z1)\n\n        #隐层第二层输出\n        z2 = tf.add(tf.matmul(a1, W2), b2)\n        # Dropout层\n        z2 = tf.nn.dropout(z2, keep_prob=KEEP_PROB)\n        a2 = tf.nn.relu(z2)\n\n        #隐层第三层输出\n        z3 = tf.add(tf.matmul(a2, W3), b3)\n        # Dropout层\n        z3 = tf.nn.dropout(z3, keep_prob=KEEP_PROB)\n        a3 = tf.nn.relu(z3)\n\n        #输出层输出\n        logits = tf.add(tf.matmul(a3, W4), b4)\n        outputs = tf.nn.sigmoid(logits, name='outputs')\n\n        #L2正则化\n        for var in [W1,W2,W3,W4]:\n            tf.add_to_collection('losses',tf.contrib.layers.l2_regularizer(L2_lamda)(var))\n\n    # loss\n    with tf.name_scope('loss'):\n        entropy_loss = tf.reduce_mean(tf.nn.sigmoid_cross_entropy_with_logits(labels=targets, logits=logits))\n        tf.add_to_collection('losses',entropy_loss)\n        loss = tf.add_n(tf.get_collection('losses'))\n    # optimizer\n    with tf.name_scope('optimizer'):\n        optimizer = tf.train.AdamOptimizer(LEARNING_RAGE).minimize(entropy_loss)\n        #optimizer = tf.train.AdamOptimizer(LEARNING_RAGE).minimize(loss) # use L2\n    # evaluation\n    with tf.name_scope('evaluation'):\n        correct_preds = tf.equal(tf.cast(tf.greater(outputs,threshold),tf.float32),targets)\n        accuracy = tf.reduce_mean(tf.reduce_sum(tf.cast(correct_preds,tf.float32),axis=1))\n\n# 训练模型\n# 存储准确率\ndnn_train_accuracy = []\ndnn_test_accuracy = []\n# 存储精确率\ndnn_train_precision = []\ndnn_test_precision = []\n# 存储召回率\ndnn_train_recall = []\ndnn_test_recall = []\n# 存储F1值\ndnn_train_F1 = []\ndnn_test_F1 = []\n\n\nwith tf.Session() as sess:\n    sess.run(tf.global_variables_initializer())\n    n_batches = int(x_train.shape[0] / BATCH_SIZE)\n\n    for epoch in range(1,EPOCHES+1):\n        total_loss = 0\n        for x_batch, y_batch in get_batch(x_train, y_train):\n            _, batch_loss = sess.run([optimizer, entropy_loss], feed_dict={inputs: x_batch, targets: y_batch})\n            total_loss += batch_loss\n            \n        # 在train 上的准确率: 随机抽取与测试数据集等量的数据进行测试\n        index = np.random.randint(0,len(x_train),len(x_test))\n        x_train_temp = x_train[index]\n        y_train_temp = y_train[index]\n        \n        train_accuracy,label_pre = sess.run([accuracy,outputs], feed_dict={inputs: x_train_temp, targets: y_train_temp})\n        dnn_train_accuracy.append(train_accuracy)\n        \n        label_pre = [int(prob[0] > threshold) for prob in label_pre] # predict label,[1,0,1,..]\n        y_train_temp = [label[0] for label in y_train_temp] # true label,[1,0,1,..]\n        \n        res_train_precision = precision_score(y_train_temp, label_pre).astype(np.float32) # train precision\n        dnn_train_precision.append(res_train_precision)\n        \n        res_train_recall = recall_score(y_train_temp, label_pre).astype(np.float32) # train recall\n        dnn_train_recall.append(res_train_recall)\n        \n        res_train_f1 = f1_score(y_train_temp, label_pre).astype(np.float32) # train F1 Score\n        dnn_train_F1.append(res_train_f1)\n\n        # 在 test上的准确率：用的是全量数据，而非批量数据\n        test_accuracy,label_pre = sess.run([accuracy,outputs], feed_dict={inputs: x_test, targets: y_test})\n        dnn_test_accuracy.append(test_accuracy)\n        \n        label_pre = [int(prob[0] > threshold) for prob in label_pre] # predict label,[1,0,1,..]\n        y_test_temp = [label[0] for label in y_test] # true label,[1,0,1,..]\n        \n        res_test_precision = precision_score(y_test_temp, label_pre).astype(np.float32)\n        dnn_test_precision.append(res_test_precision)\n        \n        res_test_recall = recall_score(y_test_temp, label_pre).astype(np.float32)\n        dnn_test_recall.append(res_test_recall)\n        \n        res_test_f1 = f1_score(y_test_temp, label_pre).astype(np.float32)\n        dnn_test_F1.append(res_test_f1)\n        \n        print('Threshold: {:.2f} , Epoch : {} , Train Loss: {:.4f} ,Train Accuracy : {:.4f} ,Test Accuracy : {:.4f} ,Train Precision : {:.4f} , Test Precision : {:.4f} , Train Recall : {:.4f} , Test Recall : {:.4f}'\n              ' ,Train F1_Score : {:.4f} ,Test F1_Score : {:.4f}'.format(threshold, epoch, total_loss / n_batches, train_accuracy, test_accuracy,res_train_precision,res_test_precision,res_train_recall,res_test_recall,res_train_f1,res_test_f1))\n    # 在 test上准确率：用的是全量数据，而非批量数据\n    test_accuracy, label_pre = sess.run([accuracy, outputs], feed_dict={inputs: x_test, targets: y_test})\n\n    # 寻找最佳阈值\n    thresholds_f1 = []\n    thresholds_precision = []\n    thresholds_recall = []\n    y_test = [label[0] for label in y_test] # true label,[1,0,1,..]\n    for thresh in np.arange(0.1, 0.501, 0.01):\n        thresh = np.round(thresh, 2)\n        label_predict = [int(prob[0] > thresh) for prob in label_pre] # predict label,[1,0,1,..]\n        res_accuracy = accuracy_score(y_test, label_predict).astype(np.float32)\n        res_f1 = f1_score(y_test, label_predict).astype(np.float32)\n        res_precision = precision_score(y_test, label_predict).astype(np.float32)\n        res_recall = recall_score(y_test, label_predict).astype(np.float32)\n        thresholds_f1.append([thresh, res_f1])\n        thresholds_precision.append([thresh, res_precision])\n        thresholds_recall.append([thresh, res_recall])\n        print(\"Epoch: %d , Threshold:%.4f , Accuracy:%.4f , Precision:%.4f , Recall:%.4f ,F1:%.4f \" % (epoch,\n                thresh, res_accuracy,res_precision, res_recall,res_f1))\n    print('*' * 100)\n    thresholds_f1.sort(key=lambda x: x[1], reverse=True)\n    best_thresh = thresholds_f1[0][0]\n    print('Best F1 Score at threshold {0} is {1}'.format(best_thresh, thresholds_f1[0][1]))\n\n    thresholds_precision.sort(key=lambda x: x[1], reverse=True)\n    best_thresh = thresholds_precision[0][0]\n    print('Best Precision at threshold {0} is {1}'.format(best_thresh, thresholds_precision[0][1]))\n\n    thresholds_recall.sort(key=lambda x: x[1], reverse=True)\n    best_thresh = thresholds_recall[0][0]\n    print('Best Recall at threshold {0} is {1}'.format(best_thresh, thresholds_recall[0][1]))\n\n    figure = plt.figure(num='The Accuracy,Precision,Recall and F1_Score of DNN Model of Threshold : %.2f'%threshold)\n    x = range(1, EPOCHES + 1)\n    \n    # 展现 train 上的 F1_Score 和 test上的 F1_Score\n    ax1 = plt.subplot(2,1,1)\n    ax1.set_title('F1_Score')\n    plt.plot(x, dnn_train_F1, label='train')\n    plt.plot(x, dnn_test_F1, label='test')\n    plt.xlim((0, EPOCHES + 1))\n    plt.ylim((0.0, 1.01))\n    plt.legend(loc='upper right')\n    \n    # 展现train上的准确率和test上的准确率\n    ax2 = plt.subplot(2,3,4)\n    ax2.set_title('Accuracy')\n    plt.plot(x, dnn_train_accuracy, label='train')\n    plt.plot(x, dnn_test_accuracy, label='test')\n    plt.xlim((0, EPOCHES + 1))\n    plt.ylim((0.0, 1.01))\n    plt.legend(loc='upper right')\n    \n    # 展现train上的精确率和test上的精确率\n    ax3 = plt.subplot(2,3,5)\n    ax3.set_title('Precision')\n    plt.plot(x, dnn_train_precision, label='train')\n    plt.plot(x, dnn_test_precision, label='test')\n    plt.legend(loc='upper right')\n    plt.xlim((0, EPOCHES + 1))\n    plt.ylim((0.0, 1.01))\n    \n    # 展现train上的召回率和test上的召回率\n    ax4 = plt.subplot(2,3,6)\n    ax4.set_title('Recall')\n    plt.plot(x, dnn_train_recall, label='train')\n    plt.plot(x, dnn_test_recall, label='test')\n    plt.xlim((0, EPOCHES + 1))\n    plt.ylim((0.0, 1.01))\n    plt.legend(loc='upper right')\n    \n    figure.subplots_adjust(hspace=0.5) # 增加子图间隔\n    figure.suptitle('The Accuracy , Precision, Recall and F1_Score of DNN Model of Threshold : %.2f' % threshold) # 大图标题\n    plt.show()\n\n    # 模型预测：在test上的准确率\n    label_pre = sess.run(outputs, feed_dict={inputs: test_questions_tokens})\n    label_pre = [int(prob[0] > threshold) for prob in label_pre]\n    # 输出预测结果\n    sub = pd.DataFrame(columns=['qid','prediction'])\n    sub['qid'] = test_qids\n    sub['prediction'] = label_pre\n    # 将question为空的qid的类别添加进去并且设置为1\n    if len(empty_question_qids) !=0:\n        empty_questions = pd.DataFrame(columns=['qid','prediction'])\n        empty_questions['qid'] = empty_question_qids\n        empty_questions['prediction'] = [1]*len(empty_question_qids)\n        sub = pd.concat([sub,empty_questions],axis=0)\n    sub.to_csv('submission.csv',index=False)\n\n    print('The total time of DNN.py program is : %d min' % ((time.time() - start)/60))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}