{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\n########################################################################################################################\n# 用神经网络中Text-CNN构建文本分类模型,具体而言分为两个部分：\n# ·数据处理\n# ·CNN模型\n########################################################################################################################\n# question统一调整为 70 个 word,不足的填充<pad>，长的则截断\n# vocabulary 大小为 <= 20万 , 且去除 词频小于1的词\n# 本版本用130万条数据来训练,Threshold:0.5,相比version3,probability > Threshold,为 1 ；probability <= Threshold , 为 0\n# 本版本做了如下改进：\n# 原数据大约有130万条,但 insincere questions只有80442条,而sincere questions却有1219559条,insincere questions 占 sincere questions的比例只有 7%\n# 将insincere questions and sincere questions 的比例设置为1:4,考虑到了样本不均衡给模型学习带来的影响\n# 对 sincere questions随机采样时,是在全量sincere questions数据集上进行的\n# 将Accuracy,Precision,Recall,F1_Score在训练过程中的变化以图表的形式展现出来了\n# 卷积核大小 filters_size = [2,3,4,5,6,7] # 2 表示filter每次覆盖的单词数，其大小为 2*embedding_size","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# Data_stage_1.py\n# -*- coding: utf-8 -*-\n# @Time    : 2019/1/7 14:56\n# @Author  : Weiyang\n\n'''\n目标：读取train.csv和test.csv数据，去除标点符号等特殊符号，将question变为word列表，单词转为小写、\n输入：train.csv,test.csv\n输出1：questions = [[word1,word2,word3,..],...]\n输出2：labels = [1,0,1,...]\n输出3：test_questions = [[word1,word2,word3,..],...]\n输出4：test_qids = [001256,05891,...]\n输出5：empty_question_qids = [] # 如果去除特殊符号后,test.csv中question为空,则question类别为1,即是insincere question;其不再参与预测\n注意:若去除特殊符号后,question为空,则将question直接删除,其不参与训练；而如果test.csv中出现此情况,则question类别确定为1,也无需参与预测;\n'''\n\nimport pandas as pd\nimport re\nimport time\nimport numpy as np\n\nstart = time.time()\n# read train.csv\ntrain_data = pd.read_csv('../input/train.csv',encoding='utf-8-sig')\n\n# 去除标点符号等特殊符号，将question变为词列表,将单词转为小写\nquestions = [] # [[word1,word2,..],..] question文本且分成一个一个词\nlabels = [] # labels\nsymbols = ['\\''] # don't, I'm\nrandint = np.random.randint(0,len(train_data['question_text']),100) # 产生100个随机整数\ncount = 0\nfor question,label in zip(train_data['question_text'],train_data['target']):\n    #if count > 30000:\n        #break\n    if count in randint:\n        print('train.csv: ',question)\n    question = re.sub(\"[\\s+\\.\\!\\/_,\\\\:;><{}\\-$%^*()+\\\"\\[\\]]+|[+——！，。：；》《？?、~@#￥%……&*（）]+\",' ',question)\n    question = [str(word) for word in question.split() if len(word.strip()) !=0 and word not in symbols]\n    question = [word.rstrip('\\'').lstrip('\\'') for word in question] # 去除左右两侧的单引号\n    question = [word.rstrip('\\\\').lstrip('\\\\') for word in question]  # 去除左右两侧的\\\\\n    question = [word.rstrip('’').lstrip('’') for word in question]  # 去除左右两侧的’\n    question = [word.rstrip('‘').lstrip('‘') for word in question]  # 去除左右两侧的‘\n    question = [word.rstrip('”').lstrip('”') for word in question]  # 去除左右两侧的”\n    question = [word.rstrip('“').lstrip('“') for word in question]  # 去除左右两侧的“\n    question = [word.rstrip('`').lstrip('`') for word in question]  # 去除左右两侧的`\n    #question = [word.lower() for word in question]  # 将单词统一为小写\n    \n    # 将question首个单词变为小写，其余单词形式不变\n    temp = []\n    for i in range(len(question)):\n        if i == 0:\n            temp.append(question[i].lower())\n        else:\n            temp.append(question[i])\n    question = temp[:]\n    \n    # 判断question是否为空\n    if len(question) == 0:\n        continue\n    questions.append(question)\n    labels.append(label)\n    if count in randint:\n        print('train.csv: ',question)\n    count += 1\n\ndel train_data\n\n# read test.csv\ntest_data = pd.read_csv('../input/test.csv',encoding='utf-8-sig')\n\n# 去除标点符号等特殊符号，将question变为词列表,将单词转为小写\ntest_questions = [] # [[word1,word2,..],..] question文本且分成一个一个词\ntest_qids = [] #qids\nempty_question_qids = [] # 如果去除特殊符号后,question为空,则question类别为1,即是insincere question;其不再参与预测\nsymbols = ['\\''] # don't,I'm,...\nrandint = np.random.randint(0,len(test_data['question_text']),100) # 产生100个随机整数\ncount = 0\nfor question,qid in zip(test_data['question_text'],test_data['qid']):\n    #if count > 3000:\n        #break\n    if count in randint:\n        print('test.csv: ',question)\n    question = re.sub(\"[\\s+\\.\\!\\/_,\\\\:;><{}\\-$%^*()+\\\"\\[\\]]+|[+——！，。：；》《？?、~@#￥%……&*（）]+\",' ',question)\n    question = [str(word) for word in question.split() if len(word.strip()) !=0 and word not in symbols]\n    question = [word.rstrip('\\'').lstrip('\\'') for word in question] # 去除左右两侧的单引号\n    question = [word.rstrip('\\\\').lstrip('\\\\') for word in question]  # 去除左右两侧的\\\\\n    question = [word.rstrip('’').lstrip('’') for word in question]  # 去除左右两侧的’\n    question = [word.rstrip('‘').lstrip('‘') for word in question]  # 去除左右两侧的‘\n    question = [word.rstrip('”').lstrip('”') for word in question]  # 去除左右两侧的”\n    question = [word.rstrip('“').lstrip('“') for word in question]  # 去除左右两侧的“\n    question = [word.rstrip('`').lstrip('`') for word in question]  # 去除左右两侧的`\n    #question = [word.lower() for word in question]  # 将单词统一为小写\n    \n    # 将question首个单词变为小写，其余单词形式不变\n    temp = []\n    for i in range(len(question)):\n        if i == 0:\n            temp.append(question[i].lower())\n        else:\n            temp.append(question[i])\n    question = temp[:]\n    \n    # 判断question是否为空\n    if len(question) == 0:\n        empty_question_qids.append(qid)\n        continue\n    test_questions.append(question)\n    test_qids.append(qid)\n    if count in randint:\n        print('test.csv: ',question)\n    count += 1\n\ndel test_data\n\nprint('The total time of Data_stage_1.py program is : %d s' %(time.time()-start))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"18e4a9aa01a3fff332a6ce18e8d2b32acae1e748"},"cell_type":"code","source":"# Data_stage_2.py\n# -*- coding: utf-8 -*- \n# @Time    : 2019/1/13 10:23 \n# @Author  : Weiyang\n\n'''\n目标：数据探索+预处理\n输入：questions,labels,test_questions\n输出1：\n'''\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nfrom collections import Counter\nimport tqdm\nimport time\n\nstart = time.time()\n\n# ------------------------------------------------数据探索：基本的统计分析------------------------------------------------\n\n'将train.csv分成insincere question 和 sincere question 两部分数据'\n# 加载insincere question\n# insincere question: 1 , sincere question: 0\ninsincere_questions = []\nsincere_questions = [] \nfor i in range(len(questions)):\n    if labels[i] == 0:\n        sincere_questions.append(questions[i])\n    elif labels[i] == 1:\n        insincere_questions.append(questions[i])\n\n# 数据探索：描述性统计\n# insincere questions and sincere questions 类别的比例\nprint('The Proportion of Insincere and sincere questions : %.2f'%(float(len(insincere_questions)/len(sincere_questions))))\n# insincere questions 统计\nprint('-' * 30 + 'Insincere Questions' + '-' * 30)\nprint('Total number of insincere questions : {}'.format(len(insincere_questions)))\nprint('The average length of insincere questions: {}'.format(np.mean([len(question) for question in insincere_questions])))\nprint('The max length of insincere questions : {}'.format(np.max([len(question) for question in insincere_questions])))\nprint('The min length of insincere questions : {}'.format(np.min([len(question) for question in insincere_questions])))\n# 统计高频词\nc = Counter([word for question in insincere_questions for word in question]).most_common(100)\nprint('Most common words in insincere questions : \\n{}'.format(c))\n\n# sincere questions 统计\nprint('-' * 30 + 'Sincere Questions' + '-' * 30)\nprint('Total number fo sincere questions : {}'.format(len(sincere_questions)))\nprint('The average length of sincere questions: {}'.format(np.mean([len(question) for question in sincere_questions])))\nprint('The max length of sincere questions : {}'.format(np.max([len(question) for question in sincere_questions])))\nprint('The min length of sincere questions : {}'.format(np.min([len(question) for question in sincere_questions])))\n# 统计高频词\nc = Counter([word for question in sincere_questions for word in question]).most_common(100)\nprint('Most common words in sincere questions : \\n{}'.format(c))\n\n# --------------------------------------------------数据预处理----------------------------------------------------------\n# ·构造词典Vocabulary\n# ·构造映射表\n# ·转换单词为tokens\n\n# 句子最大长度\nSENTENCE_LIMIT_SIZE = 70\n\n# 构造词典\n# 我们要基于整个语料来构造我们的词典，由于文本中包含许多干扰词汇，例如仅出现过1次的这类单词。对于这类极其低频词汇，我们可以对其\n# 进行去除，一方面能加快模型执行效率，一方面也能减少特殊词带来的噪声。\n\n# 将 questions+test_questions 扩展成单层列表:[word1,word2,....]\ntotal_words = [word for question in (questions+test_questions) for word in question]\n# 统计词汇\nc = Counter(total_words)\n# 倒序查看词频\nprint('The word frequency of train.csv and test.csv is : ')\nprint()\nsorted(c.most_common(),key=lambda x : x[1],reverse=True)\n\n# 初始化两个token: pad 和 unk\nvocab = ['<pad>','<unk>']\nvocab_max_length = 200000 # 词包最多能存储的单词数量\n\n# 去除出现频次大于word_frequency的单词\nword_frequency = 1\ncount = 1\nfor w,f in c.most_common():\n    if count > vocab_max_length:\n        break\n    if f > word_frequency:\n        vocab.append(w)\n    count += 1\nprint('The size of vocabulary is : {}'.format(len(vocab)))\n\n#构造映射\n# 单词到编码的映射，例如：machine ---> 10256\nword_to_token = {word: token for token,word in enumerate(vocab)}\n# 编码到单词的映射，例如：10256---> machine\ntoken_to_word = {token : word for token ,word in enumerate(vocab)}\n\n# 转换文本：对文本进行编码\ndef convert_text_to_token(sentence,word_to_token_map=word_to_token,limit_size=SENTENCE_LIMIT_SIZE):\n    '''\n    根据单词--编码映射表 将单个句子转化为token\n    :param sentence: 句子,类型是word list,[word1,word2,...]\n    :param word_to_token_map: 单词到编码的映射\n    :param limit_size: 句子最大长度。超过该长度的句子进行截断，不足的句子进行pad补全\n    :return: 句子转换为token后的列表\n    '''\n    # 获取 unknow 单词和 pad的token\n    unk_id = word_to_token_map['<unk>']\n    pad_id = word_to_token_map['<pad>']\n\n    # 对句子进行token转换，对于未在词典中出现过的词用unk的token填充\n    tokens = [word_to_token_map.get(word,unk_id) for word in sentence]\n    # 对句子长度进行规整，短的补全pad，长的截断 trunc\n    # pad\n    if len(tokens) < limit_size:\n        tokens.extend([pad_id]*(limit_size - len(tokens)))\n    # Trunc\n    else:\n        tokens = tokens[:limit_size]\n\n    return tokens\n\n# sincere questions随机采样,采样数量等于 insincere questions\nprint('Sampling sincere questions...')\nsincere_insincere_proportion = 4/1 # The ratio of sincere/insincere \nindex = np.random.randint(0,len(sincere_questions),int(len(insincere_questions)*sincere_insincere_proportion))\nsincere_questions = np.array(sincere_questions)[index]\nsincere_questions = sincere_questions.tolist()\nquestions = sincere_questions[:] + insincere_questions[:]\nlabels = [0]*len(sincere_questions) + [1]*len(insincere_questions)\ndel insincere_questions,sincere_questions\n# 混洗数据\nprint('Shuffling questions....')\nshuffled_index = np.random.permutation(range(len(labels)))\nquestions = np.array(questions)[shuffled_index]\nquestions = questions.tolist()\nlabels = np.array(labels)[shuffled_index]\nlabels = labels.tolist()\n\n# 将 questions 转为 编码列表 : [[1,5,9,55,..],....]\nprint('Begining convert questions to tokens list...')\nquestions_tokens = []\nfor question in tqdm.tqdm(questions):\n    tokens = convert_text_to_token(question)\n    questions_tokens.append(tokens)\nprint('Del questions...')\ndel questions\n# 将 test_questions 转为 编码列表 : [[1,5,9,55,..],....]\nprint('Begining convert test questions to tokens list...')\ntest_questions_tokens = []\nfor question in tqdm.tqdm(test_questions):\n    tokens = convert_text_to_token(question)\n    test_questions_tokens.append(tokens)\nprint('Del test questions...')\ndel test_questions\n\n# 转换为numpy格式,方便处理\nquestions_tokens = np.array(questions_tokens)\ntest_questions_tokens = np.array(test_questions_tokens)\nlabels = np.array(labels).reshape(-1,1)\n\nprint('The shape of all question tokens in our corpus: ({},{})'.format(*questions_tokens.shape))\nprint('The shape of all targets in our corpus : ({},)'.format(*labels.shape))\nprint('The total time of Data_stage_2.py program is : %d s' %(time.time()-start))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"783ecb3ed6e5fef8a28d364f5094065f2abcacd7"},"cell_type":"code","source":"# Data_stage_3.py\n# -*- coding: utf-8 -*- \n# @Time    : 2019/1/13 12:32 \n# @Author  : Weiyang\n\n# ------------------------------------------------构造词向量-------------------------------------------------------------\n# 这里使用 GoogleNews-vectors-negative300.bin预训练好的词向量来做embedding:\n# ·如果当前词没有对应的词向量，则用随机数产生的向量替代\n# ·如果当前词为 <PAD> ,则用 0向量替代\n\nimport warnings\nwarnings.filterwarnings(action=\"ignore\")\nimport numpy as np\nfrom gensim.models.keyedvectors import KeyedVectors\nimport time\nimport sys\n\nstart = time.time()\n\n# loads 300x1 word vectors from file.\ndef load_bin_vec(fname):\n    words_vector = KeyedVectors.load_word2vec_format(fname, binary=True) # 通过model来取对应词的词向量\n    return words_vector\n# 预训练的词向量的路径\nvectors_file =  '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\nprint('Loading all_words ....')\nprint('Loading googleNew vector....')\nwords_vector = load_bin_vec(vectors_file)  # pre-trained vectors\ngoogle_words = set() # 存储 GoogleNews-vectors-negative300 中的 词\nword_to_vec = {} # GoogleNews-vectors-negative300: {word:vector,...}\nfor word in tqdm.tqdm(words_vector.wv.vocab.keys()):\n    google_words.add(word)\n    word_to_vec[word] = np.array(words_vector[word],dtype=np.float32)     \nprint('The number of words which have pretrained-vectors in vocab is : {}'.format(len(set(vocab)&set(google_words))))\nprint()\nprint('The number of words which do not have pretrained-vectors in vocab is : {}'.format(len(set(vocab))-len(set(vocab)&set(google_words))))\ndel google_words,words_vector,vectors_file\n\n\n# 构造词向量矩阵\nVOCAB_SIZE = len(vocab) # \nEMBEDDING_SIZE = 300\n\n# 初始化词向量矩阵(这里命名为 static是因为这个词向量矩阵用预训练好的填充，无需重新训练)\nstatic_embeddings = np.zeros([VOCAB_SIZE,EMBEDDING_SIZE])\nfor word,token in tqdm.tqdm(word_to_token.items()):\n    # 用google_vector词向量填充，如果没有对应的词向量，则用随机数填充\n    word_vector = word_to_vec.get(word,0.2 * np.random.random(EMBEDDING_SIZE) - 0.1)\n    static_embeddings[token,:] = word_vector\ndel word_to_vec\n    \n# 重置PAD为0向量\npad_id = word_to_token['<pad>']\nstatic_embeddings[pad_id,:] = np.zeros(EMBEDDING_SIZE)\nstatic_embeddings = static_embeddings.astype(np.float32)\n\n# --------------------------------------------分割训练集和测试集---------------------------------------------------------\n\ndef split_train_test(x,y,train_ratio=0.8,shuffle=True):\n    '''\n    分割train 和 test\n    :param x: 输入特征序列\n    :param y: 标签序列\n    :param train_ratio: 训练样本比例\n    :param shuffle: 是否shuffle\n    :return:\n    '''\n    assert x.shape[0] == y.shape[0] , print('error shape!')\n\n    if shuffle:\n        shuffled_index = np.random.permutation(range(x.shape[0]))\n        x = x[shuffled_index]\n        y = y[shuffled_index]\n    # 分离 train 和 test\n    train_size = int(x.shape[0] * train_ratio)\n    x_train = x[:train_size]\n    x_test = x[train_size:]\n    y_train = y[:train_size]\n    y_test = y[train_size:]\n\n    return x_train,x_test,y_train,y_test\n\n# 划分train 和 test\nx_train,x_test,y_train,y_test = split_train_test(questions_tokens,labels)\nprint('Del questions_token and labels ...')\ndel questions_tokens,labels\n\n# ----------------------------------------------批量获取数据-------------------------------------------------------------\n\ndef get_batch(x,y,batch_size=300,shuffle=True):\n    assert x.shape[0] == y.shape[0],print('error shape!')\n    # shuffle\n    if shuffle:\n        shuffled_index = np.random.permutation(range(x.shape[0]))\n        x = x[shuffled_index]\n        y = y[shuffled_index]\n    # 统计共有几个完整的batch\n    n_batches = int(x.shape[0] / batch_size)\n    for i in range(n_batches - 1):\n        x_batch = x[i * batch_size:(i+1)*batch_size]\n        y_batch = y[i * batch_size:(i+1)*batch_size]\n        yield x_batch,y_batch\n        \nprint('The total time of Data_stage_3.py program is : %d s' %(time.time()-start))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42f767bfcd8576f89dc31db96861b27e104e7136"},"cell_type":"code","source":"# -*- coding: utf-8 -*- \n# @Time    : 2019/1/15 12:33 \n# @Author  : Weiyang\n\n########################################################################################################################\n# CNN 模型实现文本分类\n########################################################################################################################\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport tensorflow as tf\nfrom sklearn.metrics import precision_score\nfrom sklearn.metrics import recall_score\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import accuracy_score\nimport numpy as np\nimport time\n\n# 这里定义6种尺寸大小的filter，每种100个\nfilters_size = [2,3,4,5,6,7] # 2 表示filter每次覆盖的单词数，其大小为 2*embedding_size\nnum_filters = 100\n# 超参数\nBATCH_SIZE = 10000\nEPOCHES = 50\nLEARNING_RATE = 0.001\nL2_LAMBDA = 10\nKEEP_PROB = 0.8\nembedding_size = 300\nthreshold = 0.5 # 阈值\n\nstart = time.time()\n# 构建模型图\nwith tf.name_scope('CNN'):\n    with tf.name_scope('placeholders'):\n        inputs = tf.placeholder(dtype=tf.int32,shape=(None,SENTENCE_LIMIT_SIZE),name='inputs')\n        targets = tf.placeholder(dtype=tf.float32,shape=(None,1),name='targets')\n\n    # embeddings\n    with tf.name_scope('embeddings'):\n        #static_embeddings = tf.get_variable('embedding',[VOCAB_SIZE,EMBEDDING_SIZE]) # 不使用预训练的词向量\n        embedding_matrix = tf.Variable(initial_value=static_embeddings,trainable=False,name='embedding_matrix')\n        # embed = [None,sequence_limit_size,embedding_size]\n        embed = tf.nn.embedding_lookup(embedding_matrix,inputs,name='embed')\n        # 添加channel维度\n        # embed_expanded = [None,sequence_limit_size,embedding_size,1] ，其中1 是新增加的维度,CNN需要输入channel信息\n        embed_expanded = tf.expand_dims(embed,-1,name='embed_expand')\n\n    # 用来存储max-pooling的结果\n    pooled_outputs = []\n\n    # 迭代多个filter\n    for i,filter_size in enumerate(filters_size):\n        with tf.name_scope('conv_maxpool_%s'%filter_size):\n            filter_shape = [filter_size,embedding_size,1,num_filters]\n            W = tf.Variable(tf.truncated_normal(filter_shape,mean=0.0,stddev=0.1),name='W')\n            b = tf.Variable(tf.zeros(num_filters),name='b')\n            # conv = [num_filters*len(filters_size),sequence_limit_size-filter_size+1,1,1]\n            # 对于每个卷积核其卷积结果为[1, sequence_length - filter_size + 1, 1, 1]\n            conv = tf.nn.conv2d(input=embed_expanded,filter=W,strides=[1,1,1,1],padding='VALID',name='conv')\n            # 激活\n\n            a = tf.nn.relu(tf.nn.bias_add(conv,b),name='activations')\n            # 池化\n            max_pooling = tf.nn.max_pool(value=a,\n                                         ksize=[1,SENTENCE_LIMIT_SIZE - filter_size + 1,1,1],\n                                         strides=[1,1,1,1],\n                                         padding='VALID',\n                                         name='max_pooling')\n            pooled_outputs.append(max_pooling)\n\n    # 统计所有的filter\n    total_filters = num_filters * len(filters_size)\n    total_pool = tf.concat(pooled_outputs,3)\n    flattend_pool = tf.reshape(total_pool,(-1,total_filters))\n\n    # dropout\n    with tf.name_scope('dropout'):\n        dropout = tf.nn.dropout(flattend_pool,KEEP_PROB)\n\n    # output\n    with tf.name_scope('output'):\n        W = tf.Variable(tf.truncated_normal([total_filters,1],stddev=0.1),name='W_output')\n        b = tf.Variable(tf.zeros(1),name='b_output')\n\n        logits = tf.add(tf.matmul(dropout,W),b)\n        predictions = tf.nn.sigmoid(logits,name='predictions')\n\n    # loss\n    with tf.name_scope('loss'):\n        loss = tf.reduce_mean(tf.nn.sigmoid_cross_entropy_with_logits(labels=targets,logits=logits))\n        loss = loss + L2_LAMBDA * tf.nn.l2_loss(W)\n        optimizer = tf.train.AdamOptimizer(LEARNING_RATE).minimize(loss)\n    # evaluation\n    with tf.name_scope('evaluation'):\n        correct_preds = tf.equal(tf.cast(tf.greater(predictions,threshold),tf.float32),targets)\n        accuracy = tf.reduce_mean(tf.reduce_sum(tf.cast(correct_preds,tf.float32),axis=1))\n\n# 训练模型\n#存储训练损失\ncnn_train_loss = []\n# 存储准确率\ncnn_train_accuracy = []\ncnn_test_accuracy = []\n# 存储精确率\ncnn_train_precision = []\ncnn_test_precision = []\n# 存储召回率\ncnn_train_recall = []\ncnn_test_recall = []\n# 存储F1值\ncnn_train_F1 = []\ncnn_test_F1 = []\n\nwith tf.Session() as sess:\n    sess.run(tf.global_variables_initializer())\n    n_batches = int(x_train.shape[0] / BATCH_SIZE)\n\n    for epoch in range(1,EPOCHES+1):\n        total_loss = 0\n        for x_batch, y_batch in get_batch(x_train, y_train):\n            _, batch_loss= sess.run([optimizer,loss], feed_dict={inputs: x_batch, targets: y_batch})\n            total_loss += batch_loss\n        #存储训练损失\n        cnn_train_loss.append(total_loss/n_batches)\n        \n        # 在train 上的准确率: 随机抽取与测试数据集等量的数据进行测试\n        index = np.random.randint(0,len(x_train),len(x_test))\n        x_train_temp = x_train[index]\n        y_train_temp = y_train[index]\n        \n        # 分批次预测\n        batch_size = 20000\n        label_pre = [] # predict label,[1,0,1,..]\n        train_accuracy = []\n        for start in range(0,len(x_train_temp),batch_size):\n            train_questions = x_train_temp[start:start+batch_size]\n            train_labels = y_train_temp[start:start+batch_size]\n            accuracy_temp,label_pre_temp = sess.run([accuracy,predictions], feed_dict={inputs: train_questions,targets:train_labels})\n            train_accuracy.append(accuracy_temp)\n            label_pre_temp = [int(prob[0] > threshold) for prob in label_pre_temp]\n            label_pre.extend(label_pre_temp)\n        cnn_train_accuracy.append(np.mean(train_accuracy))\n        \n        y_train_temp = [label[0] for label in y_train_temp] # true label,[1,0,1,..]\n        \n        res_train_precision = precision_score(y_train_temp, label_pre).astype(np.float32) # train precision\n        cnn_train_precision.append(res_train_precision)\n        \n        res_train_recall = recall_score(y_train_temp, label_pre).astype(np.float32) # train recall\n        cnn_train_recall.append(res_train_recall)\n        \n        res_train_f1 = f1_score(y_train_temp, label_pre).astype(np.float32) # train F1 Score\n        cnn_train_F1.append(res_train_f1)\n\n        # 在 test上的准确率：用的是全量数据，而非批量数据\n        # 分批次预测\n        batch_size = 20000\n        label_pre = [] # predict label,[1,0,1,..]\n        label_pre_prob = [] # predict probability,[[0.5],[0.6],..]\n        test_accuracy = []\n        for start in range(0,len(x_test),batch_size):\n            test_questions = x_test[start:start+batch_size]\n            test_labels = y_test[start:start+batch_size]\n            accuracy_temp,label_pre_temp = sess.run([accuracy,predictions], feed_dict={inputs: test_questions,targets:test_labels})\n            # 最后一轮迭代,存储预测的概率以寻找最佳阈值\n            if epoch == EPOCHES:\n                label_pre_prob.extend(label_pre_temp)\n            test_accuracy.append(accuracy_temp)\n            label_pre_temp = [int(prob[0] > threshold) for prob in label_pre_temp]\n            label_pre.extend(label_pre_temp)\n        cnn_test_accuracy.append(np.mean(test_accuracy))\n        \n        y_test_temp = [label[0] for label in y_test] # true label,[1,0,1,..]\n        \n        res_test_precision = precision_score(y_test_temp, label_pre).astype(np.float32)\n        cnn_test_precision.append(res_test_precision)\n        \n        res_test_recall = recall_score(y_test_temp, label_pre).astype(np.float32)\n        cnn_test_recall.append(res_test_recall)\n        \n        res_test_f1 = f1_score(y_test_temp, label_pre).astype(np.float32)\n        cnn_test_F1.append(res_test_f1)\n        \n        print('Threshold: {:.2f} , Epoch : {} , Train Loss: {:.4f} ,Train Accuracy : {:.4f} ,Test Accuracy : {:.4f} ,Train Precision : {:.4f} , Test Precision : {:.4f} , Train Recall : {:.4f} , Test Recall : {:.4f}'\n              ' ,Train F1_Score : {:.4f} ,Test F1_Score : {:.4f}'.format(threshold, epoch, total_loss / n_batches, np.mean(train_accuracy), np.mean(test_accuracy),res_train_precision,res_test_precision,res_train_recall,res_test_recall,res_train_f1,res_test_f1))\n        \n        # 最后一轮迭代,寻找最佳阈值\n        if epoch == EPOCHES:\n             # 寻找最佳阈值\n            thresholds_f1 = []\n            thresholds_precision = []\n            thresholds_recall = []\n            y_test = [label[0] for label in y_test] # true label,[1,0,1,..]\n            for thresh in np.arange(0.1, 0.501, 0.01):\n                thresh = np.round(thresh, 2)\n                label_predict = [int(prob[0] > thresh) for prob in label_pre_prob] # predict label,[1,0,1,..]\n                res_accuracy = accuracy_score(y_test, label_predict).astype(np.float32)\n                res_f1 = f1_score(y_test, label_predict).astype(np.float32)\n                res_precision = precision_score(y_test, label_predict).astype(np.float32)\n                res_recall = recall_score(y_test, label_predict).astype(np.float32)\n                thresholds_f1.append([thresh, res_f1])\n                thresholds_precision.append([thresh, res_precision])\n                thresholds_recall.append([thresh, res_recall])\n                print(\"Epoch: %d , Threshold:%.4f , Accuracy:%.4f , Precision:%.4f , Recall:%.4f ,F1:%.4f \" % (epoch,\n                    thresh, res_accuracy,res_precision, res_recall,res_f1))\n            print('*' * 100)\n            thresholds_f1.sort(key=lambda x: x[1], reverse=True)\n            best_thresh = thresholds_f1[0][0]\n            print('Best F1 Score at threshold {0} is {1}'.format(best_thresh, thresholds_f1[0][1]))\n\n            thresholds_precision.sort(key=lambda x: x[1], reverse=True)\n            best_thresh = thresholds_precision[0][0]\n            print('Best Precision at threshold {0} is {1}'.format(best_thresh, thresholds_precision[0][1]))\n\n            thresholds_recall.sort(key=lambda x: x[1], reverse=True)\n            best_thresh = thresholds_recall[0][0]\n            print('Best Recall at threshold {0} is {1}'.format(best_thresh, thresholds_recall[0][1]))\n\n    figure = plt.figure(num='The Accuracy,Precision,Recall and F1_Score of CNN Model of Threshold : %.2f'%threshold)\n    x = range(1, EPOCHES + 1)\n    \n    # 展现 train 上的 F1_Score 和 test上的 F1_Score\n    ax1 = plt.subplot(2,1,1)\n    ax1.set_title('F1_Score')\n    plt.plot(x, cnn_train_F1, label='train')\n    plt.plot(x, cnn_test_F1, label='test')\n    plt.xlim((0, EPOCHES + 1))\n    plt.ylim((0.0, 1.01))\n    plt.legend(loc='upper right')\n    \n    # 展现train上的准确率和test上的准确率\n    ax2 = plt.subplot(2,4,5)\n    ax2.set_title('Accuracy')\n    plt.plot(x, cnn_train_accuracy, label='train')\n    plt.plot(x, cnn_test_accuracy, label='test')\n    plt.xlim((0, EPOCHES + 1))\n    plt.ylim((0.0, 1.01))\n    plt.legend(loc='upper right')\n    \n    # 展现train上的精确率和test上的精确率\n    ax3 = plt.subplot(2,4,6)\n    ax3.set_title('Precision')\n    plt.plot(x, cnn_train_precision, label='train')\n    plt.plot(x, cnn_test_precision, label='test')\n    plt.legend(loc='upper right')\n    plt.xlim((0, EPOCHES + 1))\n    plt.ylim((0.0, 1.01))\n    \n    # 展现train上的召回率和test上的召回率\n    ax4 = plt.subplot(2,4,7)\n    ax4.set_title('Recall')\n    plt.plot(x, cnn_train_recall, label='train')\n    plt.plot(x, cnn_test_recall, label='test')\n    plt.xlim((0, EPOCHES + 1))\n    plt.ylim((0.0, 1.01))\n    plt.legend(loc='upper right')\n    \n    # 展现train上的loss\n    ax5 = plt.subplot(2,4,8)\n    ax5.set_title('Train Loss')\n    plt.plot(x, cnn_train_loss, label='train')\n    plt.xlim((0, EPOCHES + 1))\n    plt.ylim((0.0, 50.0))\n    plt.legend(loc='upper right')\n    \n    figure.subplots_adjust(hspace=0.5) # 增加子图间隔\n    figure.suptitle('The Train Loss , Accuracy , Precision, Recall and F1_Score of CNN Model of Threshold : %.2f' % threshold) # 大图标题\n    plt.show()\n\n    #print(cnn_train_loss)\n    # 模型预测：在test上的准确率\n    # 分批次预测\n    batch_size = 20000\n    label_pres = []\n    for start in range(0,len(test_questions_tokens),batch_size):\n        test_questions = test_questions_tokens[start:start+batch_size]\n        label_pre = sess.run(predictions, feed_dict={inputs: test_questions})\n        label_pre = [int(prob[0] > threshold) for prob in label_pre]\n        label_pres.extend(label_pre)\n    # 输出预测结果\n    sub = pd.DataFrame(columns=['qid','prediction'])\n    sub['qid'] = test_qids\n    sub['prediction'] = label_pres\n    # 将question为空的qid的类别添加进去并且设置为1\n    if len(empty_question_qids) !=0:\n        empty_questions = pd.DataFrame(columns=['qid','prediction'])\n        empty_questions['qid'] = empty_question_qids\n        empty_questions['prediction'] = [1]*len(empty_question_qids)\n        sub = pd.concat([sub,empty_questions],axis=0)\n    sub.to_csv('submission.csv',index=False)\n    \nprint('The total time of CNN.py program is : %d min' % ((time.time() - start)/60))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}