{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T11:37:20.716943Z","iopub.execute_input":"2022-07-23T11:37:20.717885Z","iopub.status.idle":"2022-07-23T11:37:20.727668Z","shell.execute_reply.started":"2022-07-23T11:37:20.717844Z","shell.execute_reply":"2022-07-23T11:37:20.726397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:20.765374Z","iopub.execute_input":"2022-07-23T11:37:20.765749Z","iopub.status.idle":"2022-07-23T11:37:20.771806Z","shell.execute_reply.started":"2022-07-23T11:37:20.765720Z","shell.execute_reply":"2022-07-23T11:37:20.770393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/nlp-getting-started/train.csv\")\ntest = pd.read_csv(\"../input/nlp-getting-started/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:20.824944Z","iopub.execute_input":"2022-07-23T11:37:20.825495Z","iopub.status.idle":"2022-07-23T11:37:20.867567Z","shell.execute_reply.started":"2022-07-23T11:37:20.825459Z","shell.execute_reply":"2022-07-23T11:37:20.866478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 欠損値のカウント","metadata":{}},{"cell_type":"code","source":"train.isnull().sum() # 訓練データ","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:20.882428Z","iopub.execute_input":"2022-07-23T11:37:20.882842Z","iopub.status.idle":"2022-07-23T11:37:20.893981Z","shell.execute_reply.started":"2022-07-23T11:37:20.882807Z","shell.execute_reply":"2022-07-23T11:37:20.892808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isnull().sum() #テストデータ","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:20.943709Z","iopub.execute_input":"2022-07-23T11:37:20.944652Z","iopub.status.idle":"2022-07-23T11:37:20.953451Z","shell.execute_reply.started":"2022-07-23T11:37:20.944615Z","shell.execute_reply":"2022-07-23T11:37:20.952481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Training data shape: ', train.shape)\ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:21.011878Z","iopub.execute_input":"2022-07-23T11:37:21.012806Z","iopub.status.idle":"2022-07-23T11:37:21.025622Z","shell.execute_reply.started":"2022-07-23T11:37:21.012768Z","shell.execute_reply":"2022-07-23T11:37:21.024352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Target 0 : 災害との関連無、　1 : 関連有","metadata":{}},{"cell_type":"code","source":"train['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:21.071575Z","iopub.execute_input":"2022-07-23T11:37:21.072538Z","iopub.status.idle":"2022-07-23T11:37:21.080930Z","shell.execute_reply.started":"2022-07-23T11:37:21.072497Z","shell.execute_reply":"2022-07-23T11:37:21.079704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Testing data shape: ', test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:21.133588Z","iopub.execute_input":"2022-07-23T11:37:21.134404Z","iopub.status.idle":"2022-07-23T11:37:21.146843Z","shell.execute_reply.started":"2022-07-23T11:37:21.134363Z","shell.execute_reply":"2022-07-23T11:37:21.145541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ツイートの文字数(長さ)とtargetの関連性","metadata":{}},{"cell_type":"code","source":"train[\"length\"] = train[\"text\"].apply(len)#ツイートの長さ","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:21.197170Z","iopub.execute_input":"2022-07-23T11:37:21.197570Z","iopub.status.idle":"2022-07-23T11:37:21.206258Z","shell.execute_reply.started":"2022-07-23T11:37:21.197537Z","shell.execute_reply":"2022-07-23T11:37:21.205285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = \"target\",data = train,palette=\"icefire\")\nplt.title('Label Counts')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:21.257511Z","iopub.execute_input":"2022-07-23T11:37:21.258500Z","iopub.status.idle":"2022-07-23T11:37:21.411629Z","shell.execute_reply.started":"2022-07-23T11:37:21.258456Z","shell.execute_reply":"2022-07-23T11:37:21.410271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x = \"target\", y = \"length\", data = train, palette=\"icefire\")\nplt.title(\"Avg. length of each target\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:21.413637Z","iopub.execute_input":"2022-07-23T11:37:21.414429Z","iopub.status.idle":"2022-07-23T11:37:21.695949Z","shell.execute_reply.started":"2022-07-23T11:37:21.414394Z","shell.execute_reply":"2022-07-23T11:37:21.694711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(15,5))\nsns.histplot(train[train[\"target\"] == 1][\"length\"],bins = 30,ax = ax1, kde=True).set(title = \"disaster tweets\")\nsns.histplot(train[train[\"target\"] == 0][\"length\"],bins = 30,ax = ax2, kde = True).set(title = \"Not disaster tweets\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:21.697408Z","iopub.execute_input":"2022-07-23T11:37:21.697722Z","iopub.status.idle":"2022-07-23T11:37:22.155154Z","shell.execute_reply.started":"2022-07-23T11:37:21.697694Z","shell.execute_reply":"2022-07-23T11:37:22.153987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 文字数で災害の関連性有無を判断不可","metadata":{}},{"cell_type":"markdown","source":"#### keywordの確認","metadata":{}},{"cell_type":"code","source":"# target=1におけるkeyword上位\nsns.barplot(y=train[train[\"target\"] == 1][\"keyword\"].value_counts()[:20].index,x=train['keyword'].value_counts()[:20],\n            orient='h')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:22.157649Z","iopub.execute_input":"2022-07-23T11:37:22.157977Z","iopub.status.idle":"2022-07-23T11:37:22.447762Z","shell.execute_reply.started":"2022-07-23T11:37:22.157946Z","shell.execute_reply":"2022-07-23T11:37:22.446521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# target=0におけるkeyword上位\nsns.barplot(y=train[train[\"target\"] == 0][\"keyword\"].value_counts()[:20].index,x=train['keyword'].value_counts()[:20],\n            orient='h')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:22.449279Z","iopub.execute_input":"2022-07-23T11:37:22.449590Z","iopub.status.idle":"2022-07-23T11:37:22.728015Z","shell.execute_reply.started":"2022-07-23T11:37:22.449562Z","shell.execute_reply":"2022-07-23T11:37:22.726877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk #Natural Language Toolkit\nfrom nltk.corpus import stopwords\nnltk.download('stopwords')\nstopwords = stopwords.words('english')\nprint(stopwords)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:22.729618Z","iopub.execute_input":"2022-07-23T11:37:22.729951Z","iopub.status.idle":"2022-07-23T11:37:22.737488Z","shell.execute_reply.started":"2022-07-23T11:37:22.729922Z","shell.execute_reply":"2022-07-23T11:37:22.736206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import string\n\nfrom nltk.stem import WordNetLemmatizer\nimport re\nfrom nltk.corpus import stopwords #除去する単語(今回は英語(I,you,me,etc...)\n\nlemma = WordNetLemmatizer() #レンマ化(見出し語化)→カテゴリごとにグルーピング\n\n#テキスト処理する関数\ndef fix_contractions(text):\n    return contractions.fix(text)\ndef process_text(text):\n    text = re.sub(\"(@[A-Za-z0-9_]+)|([^0-9A-Za-z \\t])\", \" \",text.lower()) #文字の正規化(小文字)\n    text = text.encode(\"ascii\", \"ignore\").decode() #アスキーから変換\n    text = re.sub(r'\\n',' ', text) #改行削除\n    text = re.sub(r'<.*?>|&([a-z0-9]+|#[0-9]{1,6}|#x[0-9a-f]{1,6});', ' ', text) #HTMLタグとエンティティを削除\n    text = re.sub(r'https?:\\/\\/(www\\.)?[-a-zA-Z0–9@:%._\\+~#=]{2,256}\\.[a-z]{2,6}\\b([-a-zA-Z0–9@:%_\\+.~#?&//=]*)', '', text, flags=re.MULTILINE) #HTTPSから始まるリンクの削除\n    text = re.sub(r'[-a-zA-Z0–9@:%._\\+~#=]{2,256}\\.[a-z]{2,6}\\b([-a-zA-Z0–9@:%_\\+.~#?&//=]*)', '', text, flags=re.MULTILINE) #その他URLの削除\n    text = re.sub(\"(@[A-Za-z0-9]+)|(#[A-Za-z0-9]+)|(\\w+:\\/\\/\\S+)\",\"\",text) #メンションとハッシュタグ削除\n    text = text.replace(r'amp', ' ') #'amp'削除\n    text = text.replace(r'http', ' ') #'http'削除\n    text = text.replace(r'co', ' ') #'co'削除\n    text = re.sub(r'\\s+[a-zA-Z]\\s+', ' ', text) #単一文字削除\n    text = re.sub(r'\\b[a-zA-Z]\\b', ' ', text)\n    text = re.sub(r'\\^[a-zA-Z]\\s+', ' ', text) #最初単一文字削除\n    text = re.sub(r'\\s+[a-zA-Z]$', ' ', text) #最後単一文字削除\n    text = re.sub(r'[0-9]', ' ', text) #数字単体削除\n    text = re.sub(r'\\s+', ' ', text, flags=re.I) #複数のスペース削除\n    words = nltk.word_tokenize(text) #単語単位で分割\n    words = [lemma.lemmatize(word) for word in words if word not in set(stopwords.words(\"english\"))] \n    text = \" \".join(words) #単語間にスペース\n        \n    return text\n\ntest[\"text\"] = test[\"text\"].apply(process_text)\ntrain[\"text\"] = train[\"text\"].apply(process_text)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:22.738870Z","iopub.execute_input":"2022-07-23T11:37:22.739524Z","iopub.status.idle":"2022-07-23T11:37:46.401168Z","shell.execute_reply.started":"2022-07-23T11:37:22.739448Z","shell.execute_reply":"2022-07-23T11:37:46.399817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#訓練データから学習用とテスト用に分割\nfrom sklearn.model_selection import train_test_split\nX, y = train['keyword'], train['target']\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.1)#test:10%, , random_state=42乱数シードを固定して常に同じように分割","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:46.402762Z","iopub.execute_input":"2022-07-23T11:37:46.403087Z","iopub.status.idle":"2022-07-23T11:37:46.409971Z","shell.execute_reply.started":"2022-07-23T11:37:46.403057Z","shell.execute_reply":"2022-07-23T11:37:46.409179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 単語のベクトル化","metadata":{}},{"cell_type":"code","source":"#sklearn.feature_extraction.text import TfidfVectorizer\n#vectorizer = TfidfVectorizer(max_df=0.9, min_df=3, smooth_idf=False, ngram_range=(1,2))#各単語に加えてbi-gramも考慮\n#X_train_vec = vectorizer.fit_transform(X_train).toarray()\n#X_test_vec = vectorizer.transform(X_test).toarray()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:46.412351Z","iopub.execute_input":"2022-07-23T11:37:46.412821Z","iopub.status.idle":"2022-07-23T11:37:46.420004Z","shell.execute_reply.started":"2022-07-23T11:37:46.412793Z","shell.execute_reply":"2022-07-23T11:37:46.419264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import spacy\nnlp = spacy.load('en_core_web_lg')\nwith nlp.disable_pipes():\n    train_vecs = np.array([nlp(text).vector for text in train.text]) # doc vectors for training set\n    test_vecs = np.array([nlp(text).vector for text in test.text]) # doc vectors for testing set\nx_train, x_test, Y_train, Y_test = train_test_split(train_vecs, train.target, test_size=0.1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:37:46.421053Z","iopub.execute_input":"2022-07-23T11:37:46.421515Z","iopub.status.idle":"2022-07-23T11:39:21.835555Z","shell.execute_reply.started":"2022-07-23T11:37:46.421488Z","shell.execute_reply":"2022-07-23T11:39:21.834196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:39:21.837249Z","iopub.execute_input":"2022-07-23T11:39:21.837910Z","iopub.status.idle":"2022-07-23T11:39:21.845845Z","shell.execute_reply.started":"2022-07-23T11:39:21.837870Z","shell.execute_reply":"2022-07-23T11:39:21.844501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC #非線形SVM\nfrom sklearn.metrics import f1_score #F値\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score #正解率,適合率,再現率\n\n#clf_LR = LogisticRegression()\nclf_SVC = SVC(random_state=42, kernel='rbf', max_iter=25000)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:39:21.847466Z","iopub.execute_input":"2022-07-23T11:39:21.848549Z","iopub.status.idle":"2022-07-23T11:39:21.858031Z","shell.execute_reply.started":"2022-07-23T11:39:21.848506Z","shell.execute_reply":"2022-07-23T11:39:21.857138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_SVC.fit(x_train,Y_train)\npred = clf_SVC.predict(x_test)\n\naccu = accuracy_score(Y_test,pred)#正解率\nscore = f1_score(Y_test,pred) #予測結果のF値\n\n#モデルの評価\n\nprint(clf_SVC.__class__.__name__)\nprint(f'Accuracy: {accu}')\nprint(f'f1_score: {score}')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:39:21.859453Z","iopub.execute_input":"2022-07-23T11:39:21.860019Z","iopub.status.idle":"2022-07-23T11:39:25.916872Z","shell.execute_reply.started":"2022-07-23T11:39:21.859988Z","shell.execute_reply":"2022-07-23T11:39:25.915729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_vec = vectorizer.transform(test['text']).toarray()\n#predictions = clf.predict(test_vec)\npredictions = clf_SVC.predict(test_vecs)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:39:25.918037Z","iopub.execute_input":"2022-07-23T11:39:25.918366Z","iopub.status.idle":"2022-07-23T11:39:28.056079Z","shell.execute_reply.started":"2022-07-23T11:39:25.918339Z","shell.execute_reply":"2022-07-23T11:39:28.054911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(predictions, columns=['target'])\nsubmission['id'] = test['id']\nsubmission.set_index('id', inplace=True)\n\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:39:28.057106Z","iopub.execute_input":"2022-07-23T11:39:28.057390Z","iopub.status.idle":"2022-07-23T11:39:28.070448Z","shell.execute_reply.started":"2022-07-23T11:39:28.057366Z","shell.execute_reply":"2022-07-23T11:39:28.069396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}