{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T21:49:46.205242Z","iopub.execute_input":"2022-07-30T21:49:46.206014Z","iopub.status.idle":"2022-07-30T21:49:46.233579Z","shell.execute_reply.started":"2022-07-30T21:49:46.205905Z","shell.execute_reply":"2022-07-30T21:49:46.232607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 訓練データの読み込み\n災害ツイートデータの訓練用データを読み込む。","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/nlp-getting-started/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:49:46.616620Z","iopub.execute_input":"2022-07-30T21:49:46.616991Z","iopub.status.idle":"2022-07-30T21:49:46.663426Z","shell.execute_reply.started":"2022-07-30T21:49:46.616961Z","shell.execute_reply":"2022-07-30T21:49:46.662297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 特徴量の確認","metadata":{}},{"cell_type":"code","source":"train.describe","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-30T21:49:47.006971Z","iopub.execute_input":"2022-07-30T21:49:47.008218Z","iopub.status.idle":"2022-07-30T21:49:47.030549Z","shell.execute_reply.started":"2022-07-30T21:49:47.008156Z","shell.execute_reply":"2022-07-30T21:49:47.029628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:49:47.210349Z","iopub.execute_input":"2022-07-30T21:49:47.211351Z","iopub.status.idle":"2022-07-30T21:49:47.226790Z","shell.execute_reply.started":"2022-07-30T21:49:47.211305Z","shell.execute_reply":"2022-07-30T21:49:47.225624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:49:47.397411Z","iopub.execute_input":"2022-07-30T21:49:47.398182Z","iopub.status.idle":"2022-07-30T21:49:47.411713Z","shell.execute_reply.started":"2022-07-30T21:49:47.398139Z","shell.execute_reply":"2022-07-30T21:49:47.410562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:49:47.581993Z","iopub.execute_input":"2022-07-30T21:49:47.583033Z","iopub.status.idle":"2022-07-30T21:49:47.591676Z","shell.execute_reply.started":"2022-07-30T21:49:47.582993Z","shell.execute_reply":"2022-07-30T21:49:47.590263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# text をリストに抽出\ntext_data = train.text.tolist()\n\nword2id = {}\n\nfor line in text_data:     # 1行ずつ取得\n    for word in line.split(' '):     # 1単語ずつ取得\n        if word not in word2id and word !='\\n':     # 辞書にない単語を取得\n            id = len(word2id)\n            word2id[word] = id     # 辞書作成\n\nbow = np.zeros((len(text_data), len(word2id)))\n\nfor i, toks in enumerate(text_data):\n    for tok in toks.split(' '):\n        if tok != '\\n':\n            bow[i, word2id[tok]] += 1    # Bow作成\n\nprint(bow)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:49:47.812271Z","iopub.execute_input":"2022-07-30T21:49:47.812987Z","iopub.status.idle":"2022-07-30T21:49:48.396909Z","shell.execute_reply.started":"2022-07-30T21:49:47.812948Z","shell.execute_reply":"2022-07-30T21:49:48.395713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_words = []\nfor vec in bow:\n    num_words.append(np.sum(vec))\n\ntrain['numOfWords'] = num_words","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:49:48.399199Z","iopub.execute_input":"2022-07-30T21:49:48.399943Z","iopub.status.idle":"2022-07-30T21:49:49.109648Z","shell.execute_reply.started":"2022-07-30T21:49:48.399897Z","shell.execute_reply":"2022-07-30T21:49:49.108466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:49:49.110937Z","iopub.execute_input":"2022-07-30T21:49:49.111240Z","iopub.status.idle":"2022-07-30T21:49:49.123106Z","shell.execute_reply.started":"2022-07-30T21:49:49.111213Z","shell.execute_reply":"2022-07-30T21:49:49.121998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install stanza -U","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:49:49.125314Z","iopub.execute_input":"2022-07-30T21:49:49.125966Z","iopub.status.idle":"2022-07-30T21:50:04.701553Z","shell.execute_reply.started":"2022-07-30T21:49:49.125933Z","shell.execute_reply":"2022-07-30T21:50:04.700177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import stanza\nstanza.download('en') ","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:50:04.703605Z","iopub.execute_input":"2022-07-30T21:50:04.704155Z","iopub.status.idle":"2022-07-30T21:50:35.880423Z","shell.execute_reply.started":"2022-07-30T21:50:04.704096Z","shell.execute_reply":"2022-07-30T21:50:35.879070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nlp = stanza.Pipeline('en') # This sets up a default neural pipeline in English\n# doc = nlp(text_data[0])\n# doc.sentences[0].print_dependencies()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:50:35.881947Z","iopub.execute_input":"2022-07-30T21:50:35.882794Z","iopub.status.idle":"2022-07-30T21:50:43.639250Z","shell.execute_reply.started":"2022-07-30T21:50:35.882758Z","shell.execute_reply":"2022-07-30T21:50:43.636143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### テストデータの読み込み","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('../input/nlp-getting-started/test.csv')\n# text をリストに抽出\ntext_data_test = test.text.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:50:43.649458Z","iopub.execute_input":"2022-07-30T21:50:43.652242Z","iopub.status.idle":"2022-07-30T21:50:43.718161Z","shell.execute_reply.started":"2022-07-30T21:50:43.652077Z","shell.execute_reply":"2022-07-30T21:50:43.713767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_NUM = 7\nlocs = [0 for i in range(len(text_data_test))] # 地名\ngpes = [0 for i in range(len(text_data_test))] # 国名\ndates = [0 for i in range(len(text_data_test))] # 日にち\ntimes = [0 for i in range(len(text_data_test))] # 時間\nfacs = [0 for i in range(len(text_data_test))] # 施設名\ncards = [0 for i in range(len(text_data_test))] # 数字\nords = [0 for i in range(len(text_data_test))] # 序数\nstanza_result = [[0 for i in range(DATA_NUM)] for j in range(len(text_data_test))]\n\nfor i, line in enumerate(text_data_test):     # 1行ずつ取得\n    doc = nlp(line)\n    for sent in doc.sentences:\n        for ent in sent.ents:\n            if(ent.type == \"LOC\"):\n                locs[i] = 1\n            if(ent.type == \"GPE\"):\n                gpes[i] = 1\n            if(ent.type == \"DATE\"):\n                dates[i] = 1\n            if(ent.type == \"TIME\"):\n                times[i] = 1\n            if(ent.type == \"FAC\"):\n                facs[i] = 1\n            if(ent.type == \"CARDINAL\"):\n                cards[i] = 1\n            if(ent.type == \"ORDINAL\"):\n                ords[i] = 1\n    stanza_result[i] = [locs[i], gpes[i], dates[i], times[i], facs[i], cards[i], ords[i]]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T21:50:43.722192Z","iopub.execute_input":"2022-07-30T21:50:43.724152Z","iopub.status.idle":"2022-07-30T22:39:37.388969Z","shell.execute_reply.started":"2022-07-30T21:50:43.724004Z","shell.execute_reply":"2022-07-30T22:39:37.387694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['loc'] = locs\ntest['gpe'] = gpes\ntest['date'] = dates\ntest['time'] = times\ntest['fac'] = facs\ntest['cardinal'] = cards\ntest['ordinal'] = ords\n\ntest.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:39:37.391556Z","iopub.execute_input":"2022-07-30T22:39:37.391988Z","iopub.status.idle":"2022-07-30T22:39:37.435837Z","shell.execute_reply.started":"2022-07-30T22:39:37.391952Z","shell.execute_reply":"2022-07-30T22:39:37.434632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path ='/kaggle/working/test_stanza.txt'\ntry:\n    with open(path, 'x') as f:\n        pass\nexcept:\n    with open(path, 'w') as f:\n        pass\n\nwith open(path, 'a') as f:\n    for result in stanza_result:\n        f.write(','.join([str(_) for _ in result]))\n        f.write('\\n')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:39:37.437253Z","iopub.execute_input":"2022-07-30T22:39:37.437604Z","iopub.status.idle":"2022-07-30T22:39:37.456707Z","shell.execute_reply.started":"2022-07-30T22:39:37.437572Z","shell.execute_reply":"2022-07-30T22:39:37.455740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:39:37.457890Z","iopub.execute_input":"2022-07-30T22:39:37.458507Z","iopub.status.idle":"2022-07-30T22:39:37.478976Z","shell.execute_reply.started":"2022-07-30T22:39:37.458473Z","shell.execute_reply":"2022-07-30T22:39:37.477807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_sub = sub.target.tolist()\n\nfor i in range(len(target_sub)):\n    if locs[i] == 1 or gpes[i] == 1:\n        target = 1","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:39:37.480560Z","iopub.execute_input":"2022-07-30T22:39:37.480869Z","iopub.status.idle":"2022-07-30T22:39:37.487847Z","shell.execute_reply.started":"2022-07-30T22:39:37.480843Z","shell.execute_reply":"2022-07-30T22:39:37.486574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['target'] = target_sub\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:39:37.489477Z","iopub.execute_input":"2022-07-30T22:39:37.490667Z","iopub.status.idle":"2022-07-30T22:39:37.513756Z","shell.execute_reply.started":"2022-07-30T22:39:37.490622Z","shell.execute_reply":"2022-07-30T22:39:37.512884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub","metadata":{"execution":{"iopub.status.busy":"2022-07-30T22:39:37.515433Z","iopub.execute_input":"2022-07-30T22:39:37.516330Z","iopub.status.idle":"2022-07-30T22:39:37.528867Z","shell.execute_reply.started":"2022-07-30T22:39:37.516259Z","shell.execute_reply":"2022-07-30T22:39:37.527611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}