{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"},{"sourceId":9801,"sourceType":"datasetVersion","datasetId":6763}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nfrom tqdm import tqdm\ntqdm.pandas()\n\ntrain = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")\nprint(\"Train shape : \",train.shape)\nprint(\"Test shape : \",test.shape)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-28T05:07:21.893933Z","iopub.execute_input":"2025-08-28T05:07:21.894219Z","iopub.status.idle":"2025-08-28T05:07:29.810092Z","shell.execute_reply.started":"2025-08-28T05:07:21.894191Z","shell.execute_reply":"2025-08-28T05:07:29.809160Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"构建词汇表 单词出现次数","metadata":{}},{"cell_type":"code","source":"def build_vocab(sentences, verbose =  True):\n    vocab = {}\n    for sentence in tqdm(sentences, disable = (not verbose)):\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab\n\nsentences = train[\"question_text\"].progress_apply(lambda x: x.split()).values\nvocab = build_vocab(sentences)\nprint({k: vocab[k] for k in list(vocab)[:5]})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T05:07:43.405255Z","iopub.execute_input":"2025-08-28T05:07:43.405605Z","iopub.status.idle":"2025-08-28T05:07:55.972247Z","shell.execute_reply.started":"2025-08-28T05:07:43.405579Z","shell.execute_reply":"2025-08-28T05:07:55.970978Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"GoogleNews 加载word2vec词向量","metadata":{}},{"cell_type":"code","source":"from gensim.models import KeyedVectors\nnews_path = '/kaggle/input/googlenewsvectorsnegative300/GoogleNews-vectors-negative300.bin'\nembeddings_index = KeyedVectors.load_word2vec_format(news_path, binary=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T05:19:39.659172Z","iopub.execute_input":"2025-08-28T05:19:39.659555Z","iopub.status.idle":"2025-08-28T05:20:39.240273Z","shell.execute_reply.started":"2025-08-28T05:19:39.659528Z","shell.execute_reply":"2025-08-28T05:20:39.239521Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"词向量覆盖率检查\n\n计算并分析：数据集的词汇中有多大比例 在预训练词向量中可以找到对应的向量表示\n\n（不含词嵌入的单词个数 以及单词数目比例）\n\n如果是 OOV，Out-Of-Vocabulary words 模型的性能可能会大打折扣","metadata":{}},{"cell_type":"code","source":"import operator \n\ndef check_coverage(vocab,embeddings_index):\n    a = {}\n    oov = {}\n    k = 0\n    i = 0\n    for word in tqdm(vocab):\n        try:\n            a[word] = embeddings_index[word]\n            k += vocab[word]\n        except:\n\n            oov[word] = vocab[word]\n            i += vocab[word]\n            pass\n\n    print('Found embeddings for {:.2%} of vocab'.format(len(a) / len(vocab))) # 词汇种数比例\n    print('Found embeddings for  {:.2%} of all text'.format(k / (k + i))) # 单词个数比例\n    sorted_x = sorted(oov.items(), key=operator.itemgetter(1))[::-1]\n\n    return sorted_x\n\noov = check_coverage(vocab,embeddings_index)\noov[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T05:23:01.791134Z","iopub.execute_input":"2025-08-28T05:23:01.791511Z","iopub.status.idle":"2025-08-28T05:23:03.491918Z","shell.execute_reply.started":"2025-08-28T05:23:01.791485Z","shell.execute_reply":"2025-08-28T05:23:03.490896Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"有些标点是embedding中没有的 删除","metadata":{}},{"cell_type":"code","source":"def clean_text(x):\n    x = str(x)\n    # 1. 对 \"/-'\" 这三个符号，用空格替换\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    # 2. 对 '&' 这个符号，在其前后加上空格，使其成为独立单词\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    # 3. 对一大串标点符号，直接删除（用空字符串替换）\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x\n\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_text(x))\nsentences = train[\"question_text\"].apply(lambda x: x.split())\nvocab = build_vocab(sentences)\n\noov = check_coverage(vocab,embeddings_index)\noov[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T05:37:04.323213Z","iopub.execute_input":"2025-08-28T05:37:04.324979Z","iopub.status.idle":"2025-08-28T05:37:26.564946Z","shell.execute_reply.started":"2025-08-28T05:37:04.324933Z","shell.execute_reply":"2025-08-28T05:37:26.563536Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"删除数字","metadata":{}},{"cell_type":"code","source":"import re\n\ndef clean_numbers(x):\n\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x\n\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\nsentences = train[\"question_text\"].progress_apply(lambda x: x.split())\nvocab = build_vocab(sentences)\noov = check_coverage(vocab,embeddings_index)\noov[:20]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T05:40:44.510910Z","iopub.execute_input":"2025-08-28T05:40:44.511556Z","iopub.status.idle":"2025-08-28T05:41:12.293212Z","shell.execute_reply.started":"2025-08-28T05:40:44.511524Z","shell.execute_reply":"2025-08-28T05:41:12.292161Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"输出前四名没有实际意义的 高频出现的词 ['a','to','of','and']\n\n英式英语转美式英语：因为GoogleNews语料很可能更偏向美式英语。\n\n缩写展开：didn't -> did not，因为模型更可能学习了 did 和 not 的向量，而不是缩写形式。\n\n品牌名泛化：将特定的社交应用名称替换为更通用的“social medium”。这是一个巧妙的降维和泛化技巧，因为模型不可能包含所有新兴App的名称，但很可能有social和medium的向量。","metadata":{}},{"cell_type":"code","source":"def _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re # 把key键 也就是需要被替换的单词拼起来 方便后续检索替换\n\n\nmispell_dict = {'colour':'color',\n                'centre':'center',\n                'didnt':'did not',\n                'doesnt':'does not',\n                'isnt':'is not',\n                'shouldnt':'should not',\n                'favourite':'favorite',\n                'travelling':'traveling',\n                'counselling':'counseling',\n                'theatre':'theater',\n                'cancelled':'canceled',\n                'labour':'labor',\n                'organisation':'organization',\n                'wwii':'world war 2',\n                'citicise':'criticize',\n                \n                'instagram': 'social medium',\n                'whatsapp': 'social medium',\n                'snapchat': 'social medium'\n\n                }\nmispellings, mispellings_re = _get_mispell(mispell_dict) # 字典；需要被替换的单词拼起来\n\ndef replace_typical_misspell(text): # 替换拼写\n    def replace(match):\n        return mispellings[match.group(0)]\n\n    return mispellings_re.sub(replace, text)\n\ntrain[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x)) # 替换\nsentences = train[\"question_text\"].progress_apply(lambda x: x.split()) # 分开单词\nto_remove = ['a','to','of','and']\nsentences = [[word for word in sentence if not word in to_remove] for sentence in tqdm(sentences)]\nvocab = build_vocab(sentences)\noov = check_coverage(vocab,embeddings_index)\noov[:20]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T05:42:13.373256Z","iopub.execute_input":"2025-08-28T05:42:13.373833Z","iopub.status.idle":"2025-08-28T05:42:37.001514Z","shell.execute_reply.started":"2025-08-28T05:42:13.373799Z","shell.execute_reply":"2025-08-28T05:42:37.000524Z"}},"outputs":[],"execution_count":null}]}