{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport re\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize","metadata":{"id":"7lwnQUaUhKEg","outputId":"3b914c93-176d-4da5-f413-6b7c126e38b5","execution":{"iopub.status.busy":"2022-07-23T07:42:12.553670Z","iopub.execute_input":"2022-07-23T07:42:12.554226Z","iopub.status.idle":"2022-07-23T07:42:14.051415Z","shell.execute_reply.started":"2022-07-23T07:42:12.554114Z","shell.execute_reply":"2022-07-23T07:42:14.049265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"id":"TrxGwZI5460p","outputId":"3b570820-51bb-49b1-9a4d-c3b38024e095","execution":{"iopub.status.busy":"2022-07-23T07:42:14.057566Z","iopub.execute_input":"2022-07-23T07:42:14.059931Z","iopub.status.idle":"2022-07-23T07:42:14.072915Z","shell.execute_reply.started":"2022-07-23T07:42:14.059890Z","shell.execute_reply":"2022-07-23T07:42:14.071773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ntest = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')\nsub=pd.read_csv('/kaggle/input/nlp-getting-started/sample_submission.csv')\nprint('训练数据',train['text'].head())\nprint('测试数据',test['text'].head())","metadata":{"id":"_0DSxig5Zv3S","outputId":"195bf62d-2a15-405f-c96a-00f13fc648c2","execution":{"iopub.status.busy":"2022-07-23T07:42:14.077654Z","iopub.execute_input":"2022-07-23T07:42:14.078001Z","iopub.status.idle":"2022-07-23T07:42:14.175434Z","shell.execute_reply.started":"2022-07-23T07:42:14.077966Z","shell.execute_reply":"2022-07-23T07:42:14.174191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_text(text):\n    temp = text.lower()                                 #文档转换为小写\n    temp = re.sub('\\n', ' ' , temp)                     #删除换行符\n    temp = re.sub('\\'', '', temp)                       #删除引号\n    temp = re.sub('-', ' ', temp)                       #删除‘-’\n    temp = re.sub(r'(http|https|pic.)\\S', ' ', temp)    #删除网址\n    temp = re.sub(r'[^\\w\\s]', ' ', temp)                #删除可见及不可见符号\n    \n    return temp","metadata":{"id":"wOTKY08rZZyR","execution":{"iopub.status.busy":"2022-07-23T07:42:14.178378Z","iopub.execute_input":"2022-07-23T07:42:14.179129Z","iopub.status.idle":"2022-07-23T07:42:14.187198Z","shell.execute_reply.started":"2022-07-23T07:42:14.179091Z","shell.execute_reply":"2022-07-23T07:42:14.186199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_stopwords(text):\n    temp = [text for text in text.split() if len(text) > 3]\n    tokenized_words = word_tokenize(text)\n    temp = []\n    for word in tokenized_words:\n      if(word not in stopwords.words('english')):\n        temp.append(word)\n    temp = ' '.join(temp)\n    \n    return temp","metadata":{"id":"pO0e5EEcZgKD","execution":{"iopub.status.busy":"2022-07-23T07:42:14.188870Z","iopub.execute_input":"2022-07-23T07:42:14.189563Z","iopub.status.idle":"2022-07-23T07:42:14.199484Z","shell.execute_reply.started":"2022-07-23T07:42:14.189529Z","shell.execute_reply":"2022-07-23T07:42:14.197907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nnltk.download('punkt')\nnltk.download('stopwords')","metadata":{"id":"qmgBIu4KaWAR","outputId":"a17df16d-6a21-4698-bd6f-e0f81b15e36c","execution":{"iopub.status.busy":"2022-07-23T07:42:14.205763Z","iopub.execute_input":"2022-07-23T07:42:14.210500Z","iopub.status.idle":"2022-07-23T07:42:14.540452Z","shell.execute_reply.started":"2022-07-23T07:42:14.210459Z","shell.execute_reply":"2022-07-23T07:42:14.539511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"buffer = stopwords.words('english')\nprint(buffer)","metadata":{"id":"gE00RoKoeRwb","outputId":"96779bc8-5a18-40fc-eb67-a362bffa3ec9","execution":{"iopub.status.busy":"2022-07-23T07:42:14.544525Z","iopub.execute_input":"2022-07-23T07:42:14.547104Z","iopub.status.idle":"2022-07-23T07:42:14.559227Z","shell.execute_reply.started":"2022-07-23T07:42:14.547064Z","shell.execute_reply":"2022-07-23T07:42:14.558087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['clean'] = train['text'].apply(clean_text)\ntrain['clean'] = train['clean'].apply(remove_stopwords)\ntest['clean'] = test['text'].apply(clean_text)\ntest['clean'] = test['clean'].apply(remove_stopwords)","metadata":{"id":"UOeC-VkvZiDS","execution":{"iopub.status.busy":"2022-07-23T07:42:14.564263Z","iopub.execute_input":"2022-07-23T07:42:14.567257Z","iopub.status.idle":"2022-07-23T07:42:41.081143Z","shell.execute_reply.started":"2022-07-23T07:42:14.567217Z","shell.execute_reply":"2022-07-23T07:42:41.080201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def combine_attributes(text, keyword):\n    var_list = [text, keyword]\n    combined = ' '.join(x for x in var_list if x)\n    return combined\n\ntrain.fillna('', inplace = True)\ntrain['combine'] = train.apply(lambda x: combine_attributes(x['clean'],x['keyword']), axis = 1)\n\ntest.fillna('', inplace = True)\ntest['combine'] = test.apply(lambda x: combine_attributes(x['clean'],x['keyword']), axis = 1)","metadata":{"id":"L1rFbNK4cUmI","execution":{"iopub.status.busy":"2022-07-23T07:42:41.082363Z","iopub.execute_input":"2022-07-23T07:42:41.084294Z","iopub.status.idle":"2022-07-23T07:42:41.280021Z","shell.execute_reply.started":"2022-07-23T07:42:41.084254Z","shell.execute_reply":"2022-07-23T07:42:41.279081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"X = train['combine']\ny = train['target']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, train_size=0.7)\n\n#输出数据集大小\nprint ('原始数据集特征：',X.shape, \n       '训练数据集特征：',X_train.shape ,\n      '测试数据集特征：',X_test.shape)\n\nprint ('原始数据集标签：',y.shape, \n       '训练数据集标签：',y_train.shape ,\n      '测试数据集标签：',y_test.shape)","metadata":{"id":"ZJimvrLVc4dl"}},{"cell_type":"code","source":"X = train['combine']\ny = train['target']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, train_size=0.7)\n\n#输出数据集大小\nprint ('原始数据集特征：',X.shape, \n       '训练数据集特征：',X_train.shape ,\n      '测试数据集特征：',X_test.shape)\n\nprint ('原始数据集标签：',y.shape, \n       '训练数据集标签：',y_train.shape ,\n      '测试数据集标签：',y_test.shape)","metadata":{"id":"w0CIW-lwBsfQ","outputId":"f513fdf1-28f2-4bc9-86dd-b0c1a1009670","execution":{"iopub.status.busy":"2022-07-23T07:42:41.283506Z","iopub.execute_input":"2022-07-23T07:42:41.283883Z","iopub.status.idle":"2022-07-23T07:42:41.293548Z","shell.execute_reply.started":"2022-07-23T07:42:41.283846Z","shell.execute_reply":"2022-07-23T07:42:41.292345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\nvectorizer = TfidfVectorizer()\n\nX_train_vect = vectorizer.fit_transform(X_train)\nX_train_vect_all = vectorizer.transform(train['clean'])\nX_test_vect = vectorizer.transform(X_test)\nX_test_vect_all = vectorizer.transform(test['clean'])","metadata":{"id":"Sv3Bi9BGc8JV","execution":{"iopub.status.busy":"2022-07-23T07:42:41.295821Z","iopub.execute_input":"2022-07-23T07:42:41.296335Z","iopub.status.idle":"2022-07-23T07:42:41.615504Z","shell.execute_reply.started":"2022-07-23T07:42:41.296236Z","shell.execute_reply":"2022-07-23T07:42:41.614541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.svm import LinearSVC\n\nclf = SVC(kernel = 'linear')\nclf.fit(X_train_vect, y_train)\n\ny_pred = clf.predict(X_test_vect)","metadata":{"id":"t2IikJPQc-C8","execution":{"iopub.status.busy":"2022-07-23T07:42:41.618179Z","iopub.execute_input":"2022-07-23T07:42:41.618557Z","iopub.status.idle":"2022-07-23T07:42:45.320342Z","shell.execute_reply.started":"2022-07-23T07:42:41.618518Z","shell.execute_reply":"2022-07-23T07:42:45.319390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\naccuracy_score(y_test, y_pred)","metadata":{"id":"dWCPcyNLc_vs","outputId":"338c62dc-84e6-4b44-b26c-a3b6aab46243","execution":{"iopub.status.busy":"2022-07-23T07:42:45.321826Z","iopub.execute_input":"2022-07-23T07:42:45.322199Z","iopub.status.idle":"2022-07-23T07:42:45.331162Z","shell.execute_reply.started":"2022-07-23T07:42:45.322163Z","shell.execute_reply":"2022-07-23T07:42:45.330153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_all = clf.predict(X_test_vect_all)\nsubmission=pd.DataFrame({'id':sub['id'],'target':y_pred_all})\nsubmission.tail()\nsubmission.target.value_counts(normalize=True)\nsubmission.to_csv('./submission.csv',index=None)","metadata":{"id":"3uQoJS04dCu8","execution":{"iopub.status.busy":"2022-07-23T07:42:45.332468Z","iopub.execute_input":"2022-07-23T07:42:45.333395Z","iopub.status.idle":"2022-07-23T07:42:46.698931Z","shell.execute_reply.started":"2022-07-23T07:42:45.333358Z","shell.execute_reply":"2022-07-23T07:42:46.697736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}