{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T02:18:12.615143Z","iopub.execute_input":"2022-07-28T02:18:12.615638Z","iopub.status.idle":"2022-07-28T02:18:12.648900Z","shell.execute_reply.started":"2022-07-28T02:18:12.615544Z","shell.execute_reply":"2022-07-28T02:18:12.647771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/nlp-getting-started/train.csv')  #訓練データ\ndf_test =pd.read_csv('../input/nlp-getting-started/test.csv')     #テストデータ","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:18:14.503414Z","iopub.execute_input":"2022-07-28T02:18:14.504522Z","iopub.status.idle":"2022-07-28T02:18:14.571750Z","shell.execute_reply.started":"2022-07-28T02:18:14.504482Z","shell.execute_reply":"2022-07-28T02:18:14.570794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#訓練データの欠損値確認\nprint(df_train.isnull().sum())\n#keywordとlocationは欠損値が多い\nprint(\"\")\ndf_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:18:16.199549Z","iopub.execute_input":"2022-07-28T02:18:16.200551Z","iopub.status.idle":"2022-07-28T02:18:16.230484Z","shell.execute_reply.started":"2022-07-28T02:18:16.200515Z","shell.execute_reply":"2022-07-28T02:18:16.229033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#テストデータの欠損値確認\nprint(df_test.isnull().sum())\n#欠損値のないtextからtargetを予測する\nprint(\"\")\ndf_test.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:18:18.117994Z","iopub.execute_input":"2022-07-28T02:18:18.118622Z","iopub.status.idle":"2022-07-28T02:18:18.135982Z","shell.execute_reply.started":"2022-07-28T02:18:18.118577Z","shell.execute_reply":"2022-07-28T02:18:18.134521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train[\"target\"].value_counts())\nprint(\"\")\ndf_train[df_train[\"target\"] == 0]\n#df_train.query(\"target == 0\")\n#target=0が災害に無関係，target=1が災害に関係、無関係の方が多い","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:18:19.900693Z","iopub.execute_input":"2022-07-28T02:18:19.901395Z","iopub.status.idle":"2022-07-28T02:18:19.933657Z","shell.execute_reply.started":"2022-07-28T02:18:19.901342Z","shell.execute_reply":"2022-07-28T02:18:19.931799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 学習データ","metadata":{}},{"cell_type":"code","source":"#URLを削除\ndf_train[\"text\"] = df_train[\"text\"].str.replace(r\"https?://[a-zA-Z0-9+-=\\\\/@[\\];:,.!^'\\\"#$%&()~|`{}*<>?_]+\", \"\",regex=True)\n#df_train[df_train[\"text\"].str.contains(r\"https?://[a-zA-Z0-9+-=\\\\/@[\\];:,.!^'\\\"#$%&()~|`{}*<>?_]+\")]\n\n#HTMLのタグを削除\ndf_train[\"text\"] = df_train[\"text\"].str.replace(r\"<(\\\".*?\\\"|\\'.*?\\'|[^\\'\\\"])*?>\", \"\",regex=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:18:21.898415Z","iopub.execute_input":"2022-07-28T02:18:21.899056Z","iopub.status.idle":"2022-07-28T02:18:21.924900Z","shell.execute_reply.started":"2022-07-28T02:18:21.899021Z","shell.execute_reply":"2022-07-28T02:18:21.923635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#テキスト\ntrain_text = df_train[\"text\"].values\n#ラベル\ntrain_label = df_train[\"target\"].values","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:18:23.368191Z","iopub.execute_input":"2022-07-28T02:18:23.368595Z","iopub.status.idle":"2022-07-28T02:18:23.374544Z","shell.execute_reply.started":"2022-07-28T02:18:23.368561Z","shell.execute_reply":"2022-07-28T02:18:23.373297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#テキストは文字列だが、学習モデルの特徴量は数値にする必要あり。→単語をベクトル化\n#単語をベクトル化する際にTF-IDFを用いたベクトル化を行う\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nvectorizer = TfidfVectorizer()\ntrain_feature = vectorizer.fit_transform(train_text)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:18:25.188755Z","iopub.execute_input":"2022-07-28T02:18:25.190063Z","iopub.status.idle":"2022-07-28T02:18:25.949140Z","shell.execute_reply.started":"2022-07-28T02:18:25.190010Z","shell.execute_reply":"2022-07-28T02:18:25.948052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 学習モデル","metadata":{}},{"cell_type":"code","source":"# # k-近傍法（k-NN）\n# from sklearn.neighbors import KNeighborsClassifier\n\n# #k-NNインスタンス\n# k = int(np.sqrt(len(train_text)))\n# model = KNeighborsClassifier(n_neighbors=k)\n# #学習モデル構築。引数に訓練データの特徴量と、それに対応したラベル\n# model.fit(train_feature, train_label)\n\n# # 学習データでの正解率\n# print(\"train score:\",model.score(train_feature,train_label))\n\n# #from sklearn.metrics import accuracy_score\n# #train_predicted = model.predict(train_feature)\n# #accuracy_score(train_predicted, train_label)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # ロジスティック回帰\n# from sklearn.linear_model import LogisticRegression\n\n# #ロジスティック回帰インスタンス\n# model = LogisticRegression() \n\n# #学習モデル構築。引数に訓練データの特徴量と、それに対応したラベル\n# model.fit(train_feature, train_label)\n\n# # 学習データでの正解率\n# print(\"train score:\",model.score(train_feature,train_label))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #ナイーブベイズ\n# from sklearn.naive_bayes import MultinomialNB\n# # ナイーブベイズインスタンス\n# model = MultinomialNB()\n\n# #学習モデル構築。引数に訓練データの特徴量と、それに対応したラベル\n# model.fit(train_feature, train_label)\n\n# # 学習データでの正解率\n# print(\"train score:\",model.score(train_feature,train_label))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#サポートベクターマシン（SVM）\nfrom sklearn.svm import SVC\n\n# SVMインスタンス\nmodel = SVC()\n#学習モデル構築。引数に訓練データの特徴量と、それに対応したラベル\nmodel.fit(train_feature, train_label)\n\n# 学習データでの正解率\nprint(\"train score:\",model.score(train_feature,train_label))\n\n#kaggleに提出した中で一番スコアが高かった","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:18:45.071648Z","iopub.execute_input":"2022-07-28T02:18:45.072057Z","iopub.status.idle":"2022-07-28T02:18:59.475663Z","shell.execute_reply.started":"2022-07-28T02:18:45.072024Z","shell.execute_reply":"2022-07-28T02:18:59.474340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 線形サポートベクターマシン（SVM）\n# from sklearn.svm import LinearSVC\n\n# # SVMインスタンス\n# model = LinearSVC()\n# #学習モデル構築。引数に訓練データの特徴量と、それに対応したラベル\n# model.fit(train_feature, train_label)\n\n# # 学習データでの正解率\n# print(\"train score:\",model.score(train_feature,train_label))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## テストデータ","metadata":{}},{"cell_type":"code","source":"#URLを削除\ndf_test[\"text\"] = df_test[\"text\"].str.replace(r\"https?://[a-zA-Z0-9+-=\\\\/@[\\];:,.!^'\\\"#$%&()~|`{}*<>?_]+\", \"\",regex=True)\n#df_train[df_train[\"text\"].str.contains(r\"https?://[a-zA-Z0-9+-=\\\\/@[\\];:,.!^'\\\"#$%&()~|`{}*<>?_]+\")]\n\n#HTMLのタグを削除\ndf_test[\"text\"] = df_test[\"text\"].str.replace(r\"<(\\\".*?\\\"|\\'.*?\\'|[^\\'\\\"])*?>\", \"\",regex=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T04:21:30.711589Z","iopub.execute_input":"2022-07-19T04:21:30.711963Z","iopub.status.idle":"2022-07-19T04:21:30.732713Z","shell.execute_reply.started":"2022-07-19T04:21:30.711934Z","shell.execute_reply":"2022-07-19T04:21:30.731505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#テキスト\ntest_text = df_test[\"text\"].values","metadata":{"execution":{"iopub.status.busy":"2022-07-19T04:21:30.734366Z","iopub.execute_input":"2022-07-19T04:21:30.734710Z","iopub.status.idle":"2022-07-19T04:21:30.740582Z","shell.execute_reply.started":"2022-07-19T04:21:30.734679Z","shell.execute_reply":"2022-07-19T04:21:30.739087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#テキストは文字列だが、学習モデルの特徴量は数値にする必要あり。→単語をベクトル化\ntest_feature = vectorizer.transform(test_text)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T04:21:30.742306Z","iopub.execute_input":"2022-07-19T04:21:30.742753Z","iopub.status.idle":"2022-07-19T04:21:30.824566Z","shell.execute_reply.started":"2022-07-19T04:21:30.742712Z","shell.execute_reply":"2022-07-19T04:21:30.823664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#テストデータの予想\ntest_predicted = model.predict(test_feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T04:21:30.825847Z","iopub.execute_input":"2022-07-19T04:21:30.826360Z","iopub.status.idle":"2022-07-19T04:21:33.835188Z","shell.execute_reply.started":"2022-07-19T04:21:30.826329Z","shell.execute_reply":"2022-07-19T04:21:33.834086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#kaggleに提出\nsub = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')\nsub['target'] = list(map(int, test_predicted))\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T04:21:33.836882Z","iopub.execute_input":"2022-07-19T04:21:33.837203Z","iopub.status.idle":"2022-07-19T04:21:33.866404Z","shell.execute_reply.started":"2022-07-19T04:21:33.837176Z","shell.execute_reply":"2022-07-19T04:21:33.865357Z"},"trusted":true},"execution_count":null,"outputs":[]}]}