{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport re\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize","metadata":{"id":"7lwnQUaUhKEg","outputId":"3b914c93-176d-4da5-f413-6b7c126e38b5","execution":{"iopub.status.busy":"2022-07-27T08:01:01.954568Z","iopub.execute_input":"2022-07-27T08:01:01.955119Z","iopub.status.idle":"2022-07-27T08:01:01.962908Z","shell.execute_reply.started":"2022-07-27T08:01:01.955073Z","shell.execute_reply":"2022-07-27T08:01:01.961454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"id":"TrxGwZI5460p","outputId":"3b570820-51bb-49b1-9a4d-c3b38024e095","execution":{"iopub.status.busy":"2022-07-27T08:01:01.980570Z","iopub.execute_input":"2022-07-27T08:01:01.981322Z","iopub.status.idle":"2022-07-27T08:01:01.993686Z","shell.execute_reply.started":"2022-07-27T08:01:01.981270Z","shell.execute_reply":"2022-07-27T08:01:01.992136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ntest = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')\nsub=pd.read_csv('/kaggle/input/nlp-getting-started/sample_submission.csv')\nprint('トレーニングデータ',train['text'].head())\nprint('テストデータ',test['text'].head())","metadata":{"id":"_0DSxig5Zv3S","outputId":"195bf62d-2a15-405f-c96a-00f13fc648c2","execution":{"iopub.status.busy":"2022-07-27T08:01:02.003933Z","iopub.execute_input":"2022-07-27T08:01:02.005021Z","iopub.status.idle":"2022-07-27T08:01:02.068621Z","shell.execute_reply.started":"2022-07-27T08:01:02.004970Z","shell.execute_reply":"2022-07-27T08:01:02.067331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_text(text):\n    temp = text.lower()                                 #ドキュメントは小文字に変換されま\n    temp = re.sub('\\n', ' ' , temp)                     #改行を削除する\n    temp = re.sub('\\'', '', temp)                       #引用符を削除する\n    temp = re.sub('-', ' ', temp)                       #消去'-'\n    temp = re.sub(r'(http|https|pic.)\\S', ' ', temp)    #URLを削除\n    temp = re.sub(r'[^\\w\\s]', ' ', temp)                #表示および非表示のシンボルを削除します\n    \n    return temp","metadata":{"id":"wOTKY08rZZyR","execution":{"iopub.status.busy":"2022-07-27T08:01:02.071052Z","iopub.execute_input":"2022-07-27T08:01:02.072273Z","iopub.status.idle":"2022-07-27T08:01:02.081054Z","shell.execute_reply.started":"2022-07-27T08:01:02.072217Z","shell.execute_reply":"2022-07-27T08:01:02.079617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_stopwords(text):\n    temp = [text for text in text.split() if len(text) > 3]\n    tokenized_words = word_tokenize(text)\n    temp = []\n    for word in tokenized_words:\n      if(word not in stopwords.words('english')):\n        temp.append(word)\n    temp = ' '.join(temp)\n    \n    return temp","metadata":{"id":"pO0e5EEcZgKD","execution":{"iopub.status.busy":"2022-07-27T08:01:02.083185Z","iopub.execute_input":"2022-07-27T08:01:02.084093Z","iopub.status.idle":"2022-07-27T08:01:02.097858Z","shell.execute_reply.started":"2022-07-27T08:01:02.084042Z","shell.execute_reply":"2022-07-27T08:01:02.096788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nnltk.download('punkt')\nnltk.download('stopwords')","metadata":{"id":"qmgBIu4KaWAR","outputId":"a17df16d-6a21-4698-bd6f-e0f81b15e36c","execution":{"iopub.status.busy":"2022-07-27T08:01:02.102369Z","iopub.execute_input":"2022-07-27T08:01:02.103597Z","iopub.status.idle":"2022-07-27T08:01:42.178034Z","shell.execute_reply.started":"2022-07-27T08:01:02.103551Z","shell.execute_reply":"2022-07-27T08:01:42.176742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"buffer = stopwords.words('english')\nprint(buffer)","metadata":{"id":"gE00RoKoeRwb","outputId":"96779bc8-5a18-40fc-eb67-a362bffa3ec9","execution":{"iopub.status.busy":"2022-07-27T08:01:42.180411Z","iopub.execute_input":"2022-07-27T08:01:42.180995Z","iopub.status.idle":"2022-07-27T08:01:42.187913Z","shell.execute_reply.started":"2022-07-27T08:01:42.180928Z","shell.execute_reply":"2022-07-27T08:01:42.187021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['clean'] = train['text'].apply(clean_text)\ntrain['clean'] = train['clean'].apply(remove_stopwords)\ntest['clean'] = test['text'].apply(clean_text)\ntest['clean'] = test['clean'].apply(remove_stopwords)","metadata":{"id":"UOeC-VkvZiDS","execution":{"iopub.status.busy":"2022-07-27T08:01:42.189282Z","iopub.execute_input":"2022-07-27T08:01:42.189886Z","iopub.status.idle":"2022-07-27T08:02:08.412413Z","shell.execute_reply.started":"2022-07-27T08:01:42.189855Z","shell.execute_reply":"2022-07-27T08:02:08.411294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def combine_attributes(text, keyword):\n    var_list = [text, keyword]\n    combined = ' '.join(x for x in var_list if x)\n    return combined\n\ntrain.fillna('', inplace = True)\ntrain['combine'] = train.apply(lambda x: combine_attributes(x['clean'],x['keyword']), axis = 1)\n\ntest.fillna('', inplace = True)\ntest['combine'] = test.apply(lambda x: combine_attributes(x['clean'],x['keyword']), axis = 1)","metadata":{"id":"L1rFbNK4cUmI","execution":{"iopub.status.busy":"2022-07-27T08:02:08.415398Z","iopub.execute_input":"2022-07-27T08:02:08.415741Z","iopub.status.idle":"2022-07-27T08:02:08.650273Z","shell.execute_reply.started":"2022-07-27T08:02:08.415710Z","shell.execute_reply":"2022-07-27T08:02:08.649210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"X = train['combine']\ny = train['target']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, train_size=0.7)\n\n#输出数据集大小\nprint ('原始数据集特征：',X.shape, \n       '训练数据集特征：',X_train.shape ,\n      '测试数据集特征：',X_test.shape)\n\nprint ('原始数据集标签：',y.shape, \n       '训练数据集标签：',y_train.shape ,\n      '测试数据集标签：',y_test.shape)","metadata":{"id":"ZJimvrLVc4dl"}},{"cell_type":"code","source":"X = train['combine']\ny = train['target']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, train_size=0.7)\n\n#出力データセットサイズ\nprint ('元のデータセット機能：',X.shape, \n       'データセット機能のトレーニング：',X_train.shape ,\n      'データセット機能のテスト：',X_test.shape)\n\nprint ('元のデータセットラベル：',y.shape, \n       'トレーニングデータセットラベル：',y_train.shape ,\n      'データセットラベルをテストする：',y_test.shape)","metadata":{"id":"w0CIW-lwBsfQ","outputId":"f513fdf1-28f2-4bc9-86dd-b0c1a1009670","execution":{"iopub.status.busy":"2022-07-27T08:02:08.652225Z","iopub.execute_input":"2022-07-27T08:02:08.652708Z","iopub.status.idle":"2022-07-27T08:02:08.665111Z","shell.execute_reply.started":"2022-07-27T08:02:08.652663Z","shell.execute_reply":"2022-07-27T08:02:08.663877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\nvectorizer = TfidfVectorizer()\n\nX_train_vect = vectorizer.fit_transform(X_train)\nX_train_vect_all = vectorizer.transform(train['clean'])\nX_test_vect = vectorizer.transform(X_test)\nX_test_vect_all = vectorizer.transform(test['clean'])","metadata":{"id":"Sv3Bi9BGc8JV","execution":{"iopub.status.busy":"2022-07-27T08:02:08.666592Z","iopub.execute_input":"2022-07-27T08:02:08.667618Z","iopub.status.idle":"2022-07-27T08:02:09.063268Z","shell.execute_reply.started":"2022-07-27T08:02:08.667580Z","shell.execute_reply":"2022-07-27T08:02:09.062140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.svm import LinearSVC\n\nclf = SVC(kernel = 'linear')\nclf.fit(X_train_vect, y_train)\n\ny_pred = clf.predict(X_test_vect)","metadata":{"id":"t2IikJPQc-C8","execution":{"iopub.status.busy":"2022-07-27T08:02:09.064523Z","iopub.execute_input":"2022-07-27T08:02:09.064828Z","iopub.status.idle":"2022-07-27T08:02:13.545309Z","shell.execute_reply.started":"2022-07-27T08:02:09.064800Z","shell.execute_reply":"2022-07-27T08:02:13.543939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\naccuracy_score(y_test, y_pred)","metadata":{"id":"dWCPcyNLc_vs","outputId":"338c62dc-84e6-4b44-b26c-a3b6aab46243","execution":{"iopub.status.busy":"2022-07-27T08:02:13.547256Z","iopub.execute_input":"2022-07-27T08:02:13.547765Z","iopub.status.idle":"2022-07-27T08:02:13.560851Z","shell.execute_reply.started":"2022-07-27T08:02:13.547716Z","shell.execute_reply":"2022-07-27T08:02:13.560020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_all = clf.predict(X_test_vect_all)\nsubmission=pd.DataFrame({'id':sub['id'],'target':y_pred_all})\nsubmission.tail()\nsubmission.target.value_counts(normalize=True)\nsubmission.to_csv('./submission.csv',index=None)","metadata":{"id":"3uQoJS04dCu8","execution":{"iopub.status.busy":"2022-07-27T08:02:13.564129Z","iopub.execute_input":"2022-07-27T08:02:13.564991Z","iopub.status.idle":"2022-07-27T08:02:14.983676Z","shell.execute_reply.started":"2022-07-27T08:02:13.564929Z","shell.execute_reply":"2022-07-27T08:02:14.982401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}