{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T07:58:07.007130Z","iopub.execute_input":"2022-07-22T07:58:07.008064Z","iopub.status.idle":"2022-07-22T07:58:07.044231Z","shell.execute_reply.started":"2022-07-22T07:58:07.007945Z","shell.execute_reply":"2022-07-22T07:58:07.043350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 探索的データ解析","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom nltk.corpus import stopwords\nfrom nltk.util import ngrams\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom collections import defaultdict\nfrom collections import  Counter\nplt.style.use('ggplot')\nstop=set(stopwords.words('english'))\nimport re\nfrom nltk.tokenize import word_tokenize\nimport gensim\nimport string\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom tqdm import tqdm\nfrom keras.models import Sequential\nfrom keras.layers import Embedding,LSTM,Dense,SpatialDropout1D\nfrom keras.initializers import Constant\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:58:13.169706Z","iopub.execute_input":"2022-07-22T07:58:13.170125Z","iopub.status.idle":"2022-07-22T07:58:25.234873Z","shell.execute_reply.started":"2022-07-22T07:58:13.170093Z","shell.execute_reply":"2022-07-22T07:58:25.233953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n#os.listdir('../input/glove-global-vectors-for-word-representation/glove.6B.100d.txt')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:58:45.233383Z","iopub.execute_input":"2022-07-22T07:58:45.233769Z","iopub.status.idle":"2022-07-22T07:58:45.238338Z","shell.execute_reply.started":"2022-07-22T07:58:45.233729Z","shell.execute_reply":"2022-07-22T07:58:45.237414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet= pd.read_csv('../input/nlp-getting-started/train.csv')\ntest=pd.read_csv('../input/nlp-getting-started/test.csv')\ntweet.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:58:48.995304Z","iopub.execute_input":"2022-07-22T07:58:48.995680Z","iopub.status.idle":"2022-07-22T07:58:49.086869Z","shell.execute_reply.started":"2022-07-22T07:58:48.995652Z","shell.execute_reply":"2022-07-22T07:58:49.085866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 分布確認","metadata":{}},{"cell_type":"code","source":"x=tweet.target.value_counts()\nsns.barplot(x.index,x)\nplt.gca().set_ylabel('samples')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:58:53.535312Z","iopub.execute_input":"2022-07-22T07:58:53.535740Z","iopub.status.idle":"2022-07-22T07:58:53.752545Z","shell.execute_reply.started":"2022-07-22T07:58:53.535684Z","shell.execute_reply":"2022-07-22T07:58:53.751220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ツイートの探索的データ解析","metadata":{}},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\ntweet_len = tweet[tweet['target']==1]['text'].str.len()\nax1.hist(tweet_len,color='red')\nax1.set_title('disaster tweets')\ntweet_len=tweet[tweet['target']==0]['text'].str.len()\nax2.hist(tweet_len,color='green')\nax2.set_title('Not disaster tweets')\nfig.suptitle('Characters in tweets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:58:58.086006Z","iopub.execute_input":"2022-07-22T07:58:58.086445Z","iopub.status.idle":"2022-07-22T07:58:58.440303Z","shell.execute_reply.started":"2022-07-22T07:58:58.086408Z","shell.execute_reply":"2022-07-22T07:58:58.438953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\ntweet_len=tweet[tweet['target']==1]['text'].str.split().map(lambda x: len(x))\nax1.hist(tweet_len,color='red')\nax1.set_title('disaster tweets')\ntweet_len=tweet[tweet['target']==0]['text'].str.split().map(lambda x: len(x))\nax2.hist(tweet_len,color='green')\nax2.set_title('Not disaster tweets')\nfig.suptitle('Words in a tweet')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:59:03.073350Z","iopub.execute_input":"2022-07-22T07:59:03.073734Z","iopub.status.idle":"2022-07-22T07:59:03.447931Z","shell.execute_reply.started":"2022-07-22T07:59:03.073705Z","shell.execute_reply":"2022-07-22T07:59:03.446853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,(ax1,ax2)=plt.subplots(1,2,figsize=(10,5))\nword=tweet[tweet['target']==1]['text'].str.split().apply(lambda x : [len(i) for i in x])\nsns.distplot(word.map(lambda x: np.mean(x)),ax=ax1,color='red')\nax1.set_title('disaster')\nword=tweet[tweet['target']==0]['text'].str.split().apply(lambda x : [len(i) for i in x])\nsns.distplot(word.map(lambda x: np.mean(x)),ax=ax2,color='green')\nax2.set_title('Not disaster')\nfig.suptitle('Average word length in each tweet')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:59:07.339927Z","iopub.execute_input":"2022-07-22T07:59:07.340343Z","iopub.status.idle":"2022-07-22T07:59:08.087739Z","shell.execute_reply.started":"2022-07-22T07:59:07.340309Z","shell.execute_reply":"2022-07-22T07:59:08.086500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 単語解析","metadata":{}},{"cell_type":"code","source":"def create_corpus(target):\n    corpus=[]\n    \n    for x in tweet[tweet['target']==target]['text'].str.split():\n        for i in x:\n            corpus.append(i)\n    return corpus","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:59:11.712217Z","iopub.execute_input":"2022-07-22T07:59:11.712643Z","iopub.status.idle":"2022-07-22T07:59:11.719519Z","shell.execute_reply.started":"2022-07-22T07:59:11.712608Z","shell.execute_reply":"2022-07-22T07:59:11.718470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus = create_corpus(0)\n\ndic=defaultdict(int)\nfor word in corpus:\n    if word in stop:\n        dic[word] += 1\n\ntop = sorted(dic.items(), key=lambda x:x[1], reverse=True)[:10] \n\nx,y=zip(*top)\nplt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:59:24.979877Z","iopub.execute_input":"2022-07-22T07:59:24.980445Z","iopub.status.idle":"2022-07-22T07:59:25.445563Z","shell.execute_reply.started":"2022-07-22T07:59:24.980389Z","shell.execute_reply":"2022-07-22T07:59:25.444358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus=create_corpus(1)\n\ndic=defaultdict(int)\nfor word in corpus:\n    if word in stop:\n        dic[word]+=1\n\ntop=sorted(dic.items(), key=lambda x:x[1],reverse=True)[:10] \n\nx,y=zip(*top)\nplt.bar(x,y)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:59:28.826087Z","iopub.execute_input":"2022-07-22T07:59:28.827118Z","iopub.status.idle":"2022-07-22T07:59:29.053499Z","shell.execute_reply.started":"2022-07-22T07:59:28.827078Z","shell.execute_reply":"2022-07-22T07:59:29.052100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 句読点","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\ncorpus = create_corpus(1)\n\ndic = defaultdict(int)\nimport string\nspecial = string.punctuation\nfor i in (corpus):\n    if i in special:\n        dic[i] += 1\n        \nx,y = zip(*dic.items())\nplt.bar(x,y)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:59:40.157382Z","iopub.execute_input":"2022-07-22T07:59:40.157782Z","iopub.status.idle":"2022-07-22T07:59:40.441195Z","shell.execute_reply.started":"2022-07-22T07:59:40.157750Z","shell.execute_reply":"2022-07-22T07:59:40.440091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\ncorpus = create_corpus(0)\n\ndic = defaultdict(int)\nimport string\nspecial = string.punctuation\nfor i in (corpus):\n    if i in special:\n        dic[i] += 1\n        \nx, y = zip(*dic.items())\nplt.bar(x,y,color='green')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:59:47.003580Z","iopub.execute_input":"2022-07-22T07:59:47.004067Z","iopub.status.idle":"2022-07-22T07:59:47.307487Z","shell.execute_reply.started":"2022-07-22T07:59:47.004028Z","shell.execute_reply":"2022-07-22T07:59:47.306301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 共通する単語","metadata":{}},{"cell_type":"code","source":"counter = Counter(corpus)\nmost = counter.most_common()\nx=[]\ny=[]\nfor word,count in most[:40]:\n    if (word not in stop) :\n        x.append(word)\n        y.append(count)\n\nsns.barplot(x=y, y=x)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T07:59:58.649189Z","iopub.execute_input":"2022-07-22T07:59:58.649905Z","iopub.status.idle":"2022-07-22T07:59:58.883593Z","shell.execute_reply.started":"2022-07-22T07:59:58.649852Z","shell.execute_reply":"2022-07-22T07:59:58.882396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## N-gram解析","metadata":{}},{"cell_type":"code","source":"def get_top_tweet_bigrams(corpus, n=None):\n    vec = CountVectorizer(ngram_range=(2, 2)).fit(corpus)\n    bag_of_words = vec.transform(corpus)\n    sum_words = bag_of_words.sum(axis=0) \n    words_freq = [(word, sum_words[0, idx]) for word, idx in vec.vocabulary_.items()]\n    words_freq =sorted(words_freq, key = lambda x: x[1], reverse=True)\n    return words_freq[:n]\n\nplt.figure(figsize=(10,5))\ntop_tweet_bigrams=get_top_tweet_bigrams(tweet['text'])[:10]\nx,y=map(list,zip(*top_tweet_bigrams))\nsns.barplot(x=y,y=x)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T08:00:01.951066Z","iopub.execute_input":"2022-07-22T08:00:01.951679Z","iopub.status.idle":"2022-07-22T08:00:03.002237Z","shell.execute_reply.started":"2022-07-22T08:00:01.951632Z","shell.execute_reply":"2022-07-22T08:00:03.001354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 学習データと検証データの結合","metadata":{}},{"cell_type":"code","source":"df = pd.concat([tweet,test])\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-15T11:16:03.519145Z","iopub.execute_input":"2022-07-15T11:16:03.519939Z","iopub.status.idle":"2022-07-15T11:16:03.530838Z","shell.execute_reply.started":"2022-07-15T11:16:03.519896Z","shell.execute_reply":"2022-07-15T11:16:03.529464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## URLの排除","metadata":{}},{"cell_type":"code","source":"example = \"New competition launched :https://www.kaggle.com/c/nlp-getting-started\"\n\ndef remove_URL(text):\n    url = re.compile(r'https?://\\S+|www\\.\\S+')\n    return url.sub(r'', text)\n\nremove_URL(example)\ndf['text'] = df['text'].apply(lambda x : remove_URL(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T11:16:09.093056Z","iopub.execute_input":"2022-07-15T11:16:09.093498Z","iopub.status.idle":"2022-07-15T11:16:09.155967Z","shell.execute_reply.started":"2022-07-15T11:16:09.093461Z","shell.execute_reply":"2022-07-15T11:16:09.154574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## HTMLタグの排除","metadata":{}},{"cell_type":"code","source":"example = \"\"\"<div>\n<h1>Real or Fake</h1>\n<p>Kaggle </p>\n<a href=\"https://www.kaggle.com/c/nlp-getting-started\">getting started</a>\n</div>\"\"\"\n\ndef remove_html(text):\n    html = re.compile(r'<.*?>')\n    return html.sub(r'',text)\nprint(remove_html(example))\n\ndf['text'] = df['text'].apply(lambda x : remove_html(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T11:16:12.175470Z","iopub.execute_input":"2022-07-15T11:16:12.175915Z","iopub.status.idle":"2022-07-15T11:16:12.207278Z","shell.execute_reply.started":"2022-07-15T11:16:12.175877Z","shell.execute_reply":"2022-07-15T11:16:12.206246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 絵文字の排除","metadata":{}},{"cell_type":"code","source":"# Reference : https://gist.github.com/slowkow/7a7f61f495e3dbb7e3d767f97bd7304b\ndef remove_emoji(text):\n    emoji_pattern = re.compile(\"[\"\n                           u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                           u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                           u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                           u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           u\"\\U00002702-\\U000027B0\"\n                           u\"\\U000024C2-\\U0001F251\"\n                           \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)\n\nremove_emoji(\"Omg another Earthquake 😔😔\")\n\ndf['text'] = df['text'].apply(lambda x: remove_emoji(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T11:16:17.639172Z","iopub.execute_input":"2022-07-15T11:16:17.639613Z","iopub.status.idle":"2022-07-15T11:16:17.782588Z","shell.execute_reply.started":"2022-07-15T11:16:17.639576Z","shell.execute_reply":"2022-07-15T11:16:17.781462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 句読点の排除","metadata":{}},{"cell_type":"code","source":"def remove_punct(text):\n    table = str.maketrans('','', string.punctuation)\n    return text.translate(table)\n\nexample = \"I am a #king\"\nprint(remove_punct(example))\n\ndf['text'] = df['text'].apply(lambda x : remove_punct(x))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T11:16:20.869864Z","iopub.execute_input":"2022-07-15T11:16:20.870304Z","iopub.status.idle":"2022-07-15T11:16:20.964325Z","shell.execute_reply.started":"2022-07-15T11:16:20.870267Z","shell.execute_reply":"2022-07-15T11:16:20.963309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## スペル修正","metadata":{}},{"cell_type":"code","source":"!pip install pyspellchecker\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T11:16:23.233018Z","iopub.execute_input":"2022-07-15T11:16:23.233382Z","iopub.status.idle":"2022-07-15T11:16:38.351565Z","shell.execute_reply.started":"2022-07-15T11:16:23.233354Z","shell.execute_reply":"2022-07-15T11:16:38.350258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from spellchecker import SpellChecker\n\nspell = SpellChecker()\ndef correct_spellings(text):\n    corrected_text = []\n    misspelled_words = spell.unknown(text.split())\n    for word in text.split():\n        if word in misspelled_words:\n            corrected_text.append(spell.correction(word))\n        else:\n            corrected_text.append(word)\n    return \" \".join(corrected_text)\n        \ntext = \"corect me plese\"\ncorrect_spellings(text)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T11:16:40.648270Z","iopub.execute_input":"2022-07-15T11:16:40.649082Z","iopub.status.idle":"2022-07-15T11:16:40.836657Z","shell.execute_reply.started":"2022-07-15T11:16:40.649048Z","shell.execute_reply":"2022-07-15T11:16:40.835382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 単語分割","metadata":{}},{"cell_type":"code","source":"def create_corpus(df):\n    corpus=[]\n    for tweet in tqdm(df['text']):\n        words=[word.lower() for word in word_tokenize(tweet) if((word.isalpha()==1) & (word not in stop))]\n        corpus.append(words)\n    return corpus\n\ncorpus = create_corpus(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T11:16:45.162157Z","iopub.execute_input":"2022-07-15T11:16:45.162585Z","iopub.status.idle":"2022-07-15T11:16:47.869068Z","shell.execute_reply.started":"2022-07-15T11:16:45.162551Z","shell.execute_reply":"2022-07-15T11:16:47.867974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 単語ベクター化","metadata":{}},{"cell_type":"code","source":"embedding_dict={}\nwith open('../input/glove6b100dtxt/glove.6B.100d.txt','r') as f:\n    for line in f:\n        values = line.split()\n        word = values[0]\n        vectors = np.asarray(values[1:],'float32')\n        embedding_dict[word]  =vectors\nf.close()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T11:59:57.601575Z","iopub.execute_input":"2022-07-15T11:59:57.602006Z","iopub.status.idle":"2022-07-15T12:00:10.280447Z","shell.execute_reply.started":"2022-07-15T11:59:57.601961Z","shell.execute_reply":"2022-07-15T12:00:10.279303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_LEN=50\ntokenizer_obj=Tokenizer()\ntokenizer_obj.fit_on_texts(corpus)\nsequences=tokenizer_obj.texts_to_sequences(corpus)\n\ntweet_pad=pad_sequences(sequences,maxlen=MAX_LEN,truncating='post',padding='post')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:32:37.295961Z","iopub.execute_input":"2022-07-15T12:32:37.296381Z","iopub.status.idle":"2022-07-15T12:32:37.672787Z","shell.execute_reply.started":"2022-07-15T12:32:37.296346Z","shell.execute_reply":"2022-07-15T12:32:37.671474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_index=tokenizer_obj.word_index\nprint('Number of unique words:',len(word_index))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:32:39.797745Z","iopub.execute_input":"2022-07-15T12:32:39.798255Z","iopub.status.idle":"2022-07-15T12:32:39.809593Z","shell.execute_reply.started":"2022-07-15T12:32:39.798211Z","shell.execute_reply":"2022-07-15T12:32:39.807530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 単語ベクター配列作成","metadata":{}},{"cell_type":"code","source":"num_words = len(word_index) + 1\nembedding_matrix = np.zeros((num_words, 100))\n\nfor word,i in tqdm(word_index.items()):\n    if i > num_words:\n        continue\n    \n    emb_vec = embedding_dict.get(word)\n    if emb_vec is not None:\n        embedding_matrix[i] = emb_vec\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:32:41.911910Z","iopub.execute_input":"2022-07-15T12:32:41.912607Z","iopub.status.idle":"2022-07-15T12:32:41.988388Z","shell.execute_reply.started":"2022-07-15T12:32:41.912565Z","shell.execute_reply":"2022-07-15T12:32:41.987035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Baselineモデルの準備","metadata":{}},{"cell_type":"code","source":"model = Sequential()\n\nembedding=Embedding(num_words, 100, embeddings_initializer=Constant(embedding_matrix),\n                   input_length=MAX_LEN, trainable=False)\n\nmodel.add(embedding)\nmodel.add(SpatialDropout1D(0.2))\nmodel.add(LSTM(64, dropout=0.2, recurrent_dropout=0.2))\nmodel.add(Dense(1, activation='sigmoid'))\n\noptimzer = Adam(learning_rate=1e-5)\n\nmodel.compile(loss='binary_crossentropy',optimizer=optimzer,metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:32:44.901636Z","iopub.execute_input":"2022-07-15T12:32:44.902718Z","iopub.status.idle":"2022-07-15T12:32:45.262794Z","shell.execute_reply.started":"2022-07-15T12:32:44.902676Z","shell.execute_reply":"2022-07-15T12:32:45.261391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## データ分割","metadata":{}},{"cell_type":"code","source":"train=tweet_pad[:tweet.shape[0]]\ntest=tweet_pad[tweet.shape[0]:]\n\nX_train,X_test,y_train,y_test = train_test_split(train,tweet['target'].values,test_size=0.15)\nprint('Shape of train',X_train.shape)\nprint(\"Shape of Validation \",X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:32:49.701030Z","iopub.execute_input":"2022-07-15T12:32:49.701722Z","iopub.status.idle":"2022-07-15T12:32:49.713464Z","shell.execute_reply.started":"2022-07-15T12:32:49.701676Z","shell.execute_reply":"2022-07-15T12:32:49.712386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 学習","metadata":{}},{"cell_type":"code","source":"history=model.fit(X_train,y_train,batch_size=4,epochs=15,validation_data=(X_test,y_test),verbose=2)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:32:53.094231Z","iopub.execute_input":"2022-07-15T12:32:53.095476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 提出用ファイルの作成","metadata":{}},{"cell_type":"code","source":"sample_sub=pd.read_csv('../input/nlp-getting-started/sample_submission.csv')\n\ny_pre = model.predict(test)\ny_pre = np.round(y_pre).astype(int).reshape(3263)\nsub = pd.DataFrame({'id':sample_sub['id'].values.tolist(), 'target':y_pre})\nsub.to_csv('submission.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}